diff --git a/CHANGELOG.md b/CHANGELOG.md index 99865aa..15f5d7f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,9 @@ ### Added +- Auto/Fast/Efficient/Powersave protocol profiles with backend-selected threads + by default, also applied during mapped-speaker regeneration. + - Versioned UTF-8 YAML import/export for the SQLite terminology glossary, with stable entry IDs, complete validation, and atomic replacement semantics. - First Streamlit MVP for audio upload, Meeting Context entry, participant diff --git a/PROJECT_KNOWLEDGE.md b/PROJECT_KNOWLEDGE.md index c7bcc1f..594d415 100644 --- a/PROJECT_KNOWLEDGE.md +++ b/PROJECT_KNOWLEDGE.md @@ -95,6 +95,14 @@ PyTorch/ROCm processed the same duration in approximately 98.6 seconds. These are validation observations, not performance guarantees or hardware requirements. CPU execution remains supported and may be substantially slower. +Protocol-generation performance is selected through intentionally abstract +profiles rather than hardware controls in the GUI. `auto` is the default and +delegates thread selection to the inference backend; with Ollama this means +omitting `num_thread` entirely. The explicit resource profiles currently map +`fast` to 16 Ollama CPU threads, `efficient` to 10, and `powersave` to 4. These +concrete mappings belong to the backend/configuration boundary and may evolve +independently of the user-facing profile semantics. + ## Meeting Context and Speakers `MeetingContext` is a structured domain object containing meeting metadata, diff --git a/README.md b/README.md index 7f32915..e29bf63 100644 --- a/README.md +++ b/README.md @@ -82,6 +82,14 @@ Audio normalization is enabled by default and can be disabled in the processing options. This controls loudness normalization only: Meeting Lab still prepares every WAV, FLAC or M4A source as canonical audio before transcription. +The **Performance profile** selector controls protocol-generation runtime using +the abstract Auto (default), Fast, Efficient, and Powersave profiles. Auto lets +the inference backend select its own thread configuration; for Ollama, Meeting +Assistant intentionally sends no `num_thread` option. Fast, Efficient, and +Powersave are explicit resource profiles currently mapped to 16, 10, and 4 +Ollama CPU threads. These concrete mappings may evolve independently of the UI +semantics. + The People section can export its current entries to a UTF-8 `people.yaml` file and replace them from a previous `.yaml` or `.yml` export. Stable person IDs, names, roles, organizations and attendance states are retained. This is a small diff --git a/src/mka/application/meeting_service.py b/src/mka/application/meeting_service.py index bb506f0..851add6 100644 --- a/src/mka/application/meeting_service.py +++ b/src/mka/application/meeting_service.py @@ -16,6 +16,7 @@ import yaml from mka.application.config import AppSettings from mka.application.glossary import GlossaryRepository, render_glossary_terms +from mka.application.performance import DEFAULT_PERFORMANCE_PROFILE, resolve_ollama_num_thread STAGES = ("preparing", "transcription", "diarization", "protocol_generation") @@ -65,6 +66,7 @@ class ParticipantInput: class ProcessingOptions: diarization_enabled: bool = False audio_normalization: bool = True + performance_profile: str = DEFAULT_PERFORMANCE_PROFILE @dataclass(frozen=True) @@ -229,6 +231,7 @@ class MeetingProcessingService: "model": self.settings.protocol_model, "ollama_endpoint": self.settings.ollama_endpoint, "protocol_num_ctx": self.settings.protocol_num_ctx, + "protocol_num_thread": resolve_ollama_num_thread(options.performance_profile), "protocol_safe_input_token_budget": ( self.settings.protocol_safe_input_token_budget ), @@ -357,6 +360,7 @@ class MeetingProcessingService: run_dir: Path, speaker_mappings: dict[str, str], progress_sink: Callable[[AppProgressEvent], None] | None = None, + performance_profile: str = DEFAULT_PERFORMANCE_PROFILE, ) -> ProcessingOutcome: """Persist confirmed mappings and regenerate only the direct protocol.""" review = self.load_speaker_mapping_review(run_dir) @@ -401,6 +405,7 @@ class MeetingProcessingService: model=self.settings.protocol_model, ollama_endpoint=self.settings.ollama_endpoint, protocol_num_ctx=self.settings.protocol_num_ctx, + protocol_num_thread=resolve_ollama_num_thread(performance_profile), protocol_safe_input_token_budget=(self.settings.protocol_safe_input_token_budget), ) protocol_path = Path(result.protocol_path) diff --git a/src/mka/application/performance.py b/src/mka/application/performance.py new file mode 100644 index 0000000..e828856 --- /dev/null +++ b/src/mka/application/performance.py @@ -0,0 +1,23 @@ +"""Abstract performance profiles resolved to current backend runtime options.""" + +from __future__ import annotations + +DEFAULT_PERFORMANCE_PROFILE = "auto" +PERFORMANCE_PROFILES = ("auto", "fast", "efficient", "powersave") + +_OLLAMA_THREADS_BY_PROFILE = { + "fast": 16, + "efficient": 10, + "powersave": 4, +} + + +def resolve_ollama_num_thread(profile: str) -> int | None: + """Resolve a profile, leaving automatic thread selection to Ollama for Auto.""" + if profile == "auto": + return None + try: + return _OLLAMA_THREADS_BY_PROFILE[profile] + except KeyError as exc: + allowed = ", ".join(PERFORMANCE_PROFILES) + raise ValueError(f"Unknown performance profile {profile!r}; expected: {allowed}.") from exc diff --git a/src/mka/ui/streamlit_app.py b/src/mka/ui/streamlit_app.py index 4fadadb..e713e4e 100644 --- a/src/mka/ui/streamlit_app.py +++ b/src/mka/ui/streamlit_app.py @@ -34,6 +34,7 @@ from mka.application.people_yaml import ( export_people_yaml, import_people_yaml, ) +from mka.application.performance import DEFAULT_PERFORMANCE_PROFILE, PERFORMANCE_PROFILES from mka.application.run_inputs import ( RunInputJsonError, RunInputState, @@ -495,7 +496,13 @@ def _render_result() -> None: ): try: with st.spinner("Regenerating protocol without rerunning audio processing..."): - regenerated = service.regenerate_protocol(outcome.run_dir, selections) + regenerated = service.regenerate_protocol( + outcome.run_dir, + selections, + performance_profile=st.session_state.get( + "performance_profile", DEFAULT_PERFORMANCE_PROFILE + ), + ) except (OSError, RuntimeError, ValueError) as exc: st.error(f"Protocol regeneration failed: {exc}") else: @@ -551,6 +558,13 @@ def main() -> None: participants = _render_participants() st.header("Processing options") + performance_profile = st.selectbox( + "Performance profile", + options=PERFORMANCE_PROFILES, + format_func=str.title, + key="performance_profile", + help="Controls protocol-generation performance using an abstract runtime profile.", + ) audio_normalization = st.checkbox( "Audio normalization", value=True, @@ -619,6 +633,7 @@ def main() -> None: ProcessingOptions( diarization_enabled=diarization_enabled, audio_normalization=audio_normalization, + performance_profile=performance_profile, ), progress_sink=callback, ) diff --git a/tests/test_meeting_service.py b/tests/test_meeting_service.py index 736de78..c7dfaa7 100644 --- a/tests/test_meeting_service.py +++ b/tests/test_meeting_service.py @@ -561,3 +561,50 @@ def test_uploaded_source_is_preserved_in_meeting_directory(tmp_path: Path) -> No assert destination.parent == tmp_path / "meetings" / "meeting-1" / "uploads" assert destination.name.endswith("_unsafe.wav") assert destination.read_bytes() == b"source audio" + + +@pytest.mark.parametrize( + ("profile", "threads"), + [("fast", 16), ("efficient", 10), ("powersave", 4)], +) +def test_process_resolves_profile_for_protocol_generation( + tmp_path: Path, profile: str, threads: int +) -> None: + service, gateway = make_service(tmp_path) + audio = tmp_path / "meeting.wav" + audio.write_bytes(b"audio") + + service.process( + audio, + meeting(), + participants(), + ProcessingOptions(performance_profile=profile), + ) + + assert gateway.config_values is not None + assert gateway.config_values["protocol_num_thread"] == threads + + +def test_process_defaults_to_backend_thread_selection(tmp_path: Path) -> None: + service, gateway = make_service(tmp_path) + audio = tmp_path / "meeting.wav" + audio.write_bytes(b"audio") + + service.process(audio, meeting(), participants(), ProcessingOptions()) + + assert gateway.config_values is not None + assert gateway.config_values["protocol_num_thread"] is None + + +def test_protocol_regeneration_propagates_explicit_profile(tmp_path: Path) -> None: + service, gateway = make_service(tmp_path) + write_speaker_review_artifacts(gateway.run_dir) + + service.regenerate_protocol( + gateway.run_dir, + {}, + performance_profile="fast", + ) + + assert gateway.regeneration is not None + assert gateway.regeneration["options"]["protocol_num_thread"] == 16 diff --git a/tests/test_performance.py b/tests/test_performance.py new file mode 100644 index 0000000..7f3ea48 --- /dev/null +++ b/tests/test_performance.py @@ -0,0 +1,21 @@ +import pytest + +from mka.application.meeting_service import ProcessingOptions +from mka.application.performance import PERFORMANCE_PROFILES, resolve_ollama_num_thread + + +def test_default_profile_is_auto() -> None: + assert ProcessingOptions().performance_profile == "auto" + assert PERFORMANCE_PROFILES[0] == "auto" + + +def test_auto_profile_has_no_ollama_thread_override() -> None: + assert resolve_ollama_num_thread("auto") is None + + +@pytest.mark.parametrize( + ("profile", "threads"), + [("fast", 16), ("efficient", 10), ("powersave", 4)], +) +def test_profiles_resolve_to_current_ollama_thread_counts(profile: str, threads: int) -> None: + assert resolve_ollama_num_thread(profile) == threads