Add protocol performance profiles

This commit is contained in:
2026-09-12 11:50:30 +02:00
parent e2a0434941
commit 77be2b39ff
8 changed files with 131 additions and 1 deletions
+3
View File
@@ -4,6 +4,9 @@
### Added ### Added
- Auto/Fast/Efficient/Powersave protocol profiles with backend-selected threads
by default, also applied during mapped-speaker regeneration.
- Versioned UTF-8 YAML import/export for the SQLite terminology glossary, with - Versioned UTF-8 YAML import/export for the SQLite terminology glossary, with
stable entry IDs, complete validation, and atomic replacement semantics. stable entry IDs, complete validation, and atomic replacement semantics.
- First Streamlit MVP for audio upload, Meeting Context entry, participant - First Streamlit MVP for audio upload, Meeting Context entry, participant
+8
View File
@@ -95,6 +95,14 @@ PyTorch/ROCm processed the same duration in approximately 98.6 seconds. These
are validation observations, not performance guarantees or hardware are validation observations, not performance guarantees or hardware
requirements. CPU execution remains supported and may be substantially slower. requirements. CPU execution remains supported and may be substantially slower.
Protocol-generation performance is selected through intentionally abstract
profiles rather than hardware controls in the GUI. `auto` is the default and
delegates thread selection to the inference backend; with Ollama this means
omitting `num_thread` entirely. The explicit resource profiles currently map
`fast` to 16 Ollama CPU threads, `efficient` to 10, and `powersave` to 4. These
concrete mappings belong to the backend/configuration boundary and may evolve
independently of the user-facing profile semantics.
## Meeting Context and Speakers ## Meeting Context and Speakers
`MeetingContext` is a structured domain object containing meeting metadata, `MeetingContext` is a structured domain object containing meeting metadata,
+8
View File
@@ -82,6 +82,14 @@ Audio normalization is enabled by default and can be disabled in the processing
options. This controls loudness normalization only: Meeting Lab still prepares options. This controls loudness normalization only: Meeting Lab still prepares
every WAV, FLAC or M4A source as canonical audio before transcription. every WAV, FLAC or M4A source as canonical audio before transcription.
The **Performance profile** selector controls protocol-generation runtime using
the abstract Auto (default), Fast, Efficient, and Powersave profiles. Auto lets
the inference backend select its own thread configuration; for Ollama, Meeting
Assistant intentionally sends no `num_thread` option. Fast, Efficient, and
Powersave are explicit resource profiles currently mapped to 16, 10, and 4
Ollama CPU threads. These concrete mappings may evolve independently of the UI
semantics.
The People section can export its current entries to a UTF-8 `people.yaml` file The People section can export its current entries to a UTF-8 `people.yaml` file
and replace them from a previous `.yaml` or `.yml` export. Stable person IDs, and replace them from a previous `.yaml` or `.yml` export. Stable person IDs,
names, roles, organizations and attendance states are retained. This is a small names, roles, organizations and attendance states are retained. This is a small
+5
View File
@@ -16,6 +16,7 @@ import yaml
from mka.application.config import AppSettings from mka.application.config import AppSettings
from mka.application.glossary import GlossaryRepository, render_glossary_terms from mka.application.glossary import GlossaryRepository, render_glossary_terms
from mka.application.performance import DEFAULT_PERFORMANCE_PROFILE, resolve_ollama_num_thread
STAGES = ("preparing", "transcription", "diarization", "protocol_generation") STAGES = ("preparing", "transcription", "diarization", "protocol_generation")
@@ -65,6 +66,7 @@ class ParticipantInput:
class ProcessingOptions: class ProcessingOptions:
diarization_enabled: bool = False diarization_enabled: bool = False
audio_normalization: bool = True audio_normalization: bool = True
performance_profile: str = DEFAULT_PERFORMANCE_PROFILE
@dataclass(frozen=True) @dataclass(frozen=True)
@@ -229,6 +231,7 @@ class MeetingProcessingService:
"model": self.settings.protocol_model, "model": self.settings.protocol_model,
"ollama_endpoint": self.settings.ollama_endpoint, "ollama_endpoint": self.settings.ollama_endpoint,
"protocol_num_ctx": self.settings.protocol_num_ctx, "protocol_num_ctx": self.settings.protocol_num_ctx,
"protocol_num_thread": resolve_ollama_num_thread(options.performance_profile),
"protocol_safe_input_token_budget": ( "protocol_safe_input_token_budget": (
self.settings.protocol_safe_input_token_budget self.settings.protocol_safe_input_token_budget
), ),
@@ -357,6 +360,7 @@ class MeetingProcessingService:
run_dir: Path, run_dir: Path,
speaker_mappings: dict[str, str], speaker_mappings: dict[str, str],
progress_sink: Callable[[AppProgressEvent], None] | None = None, progress_sink: Callable[[AppProgressEvent], None] | None = None,
performance_profile: str = DEFAULT_PERFORMANCE_PROFILE,
) -> ProcessingOutcome: ) -> ProcessingOutcome:
"""Persist confirmed mappings and regenerate only the direct protocol.""" """Persist confirmed mappings and regenerate only the direct protocol."""
review = self.load_speaker_mapping_review(run_dir) review = self.load_speaker_mapping_review(run_dir)
@@ -401,6 +405,7 @@ class MeetingProcessingService:
model=self.settings.protocol_model, model=self.settings.protocol_model,
ollama_endpoint=self.settings.ollama_endpoint, ollama_endpoint=self.settings.ollama_endpoint,
protocol_num_ctx=self.settings.protocol_num_ctx, protocol_num_ctx=self.settings.protocol_num_ctx,
protocol_num_thread=resolve_ollama_num_thread(performance_profile),
protocol_safe_input_token_budget=(self.settings.protocol_safe_input_token_budget), protocol_safe_input_token_budget=(self.settings.protocol_safe_input_token_budget),
) )
protocol_path = Path(result.protocol_path) protocol_path = Path(result.protocol_path)
+23
View File
@@ -0,0 +1,23 @@
"""Abstract performance profiles resolved to current backend runtime options."""
from __future__ import annotations
DEFAULT_PERFORMANCE_PROFILE = "auto"
PERFORMANCE_PROFILES = ("auto", "fast", "efficient", "powersave")
_OLLAMA_THREADS_BY_PROFILE = {
"fast": 16,
"efficient": 10,
"powersave": 4,
}
def resolve_ollama_num_thread(profile: str) -> int | None:
"""Resolve a profile, leaving automatic thread selection to Ollama for Auto."""
if profile == "auto":
return None
try:
return _OLLAMA_THREADS_BY_PROFILE[profile]
except KeyError as exc:
allowed = ", ".join(PERFORMANCE_PROFILES)
raise ValueError(f"Unknown performance profile {profile!r}; expected: {allowed}.") from exc
+16 -1
View File
@@ -34,6 +34,7 @@ from mka.application.people_yaml import (
export_people_yaml, export_people_yaml,
import_people_yaml, import_people_yaml,
) )
from mka.application.performance import DEFAULT_PERFORMANCE_PROFILE, PERFORMANCE_PROFILES
from mka.application.run_inputs import ( from mka.application.run_inputs import (
RunInputJsonError, RunInputJsonError,
RunInputState, RunInputState,
@@ -495,7 +496,13 @@ def _render_result() -> None:
): ):
try: try:
with st.spinner("Regenerating protocol without rerunning audio processing..."): with st.spinner("Regenerating protocol without rerunning audio processing..."):
regenerated = service.regenerate_protocol(outcome.run_dir, selections) regenerated = service.regenerate_protocol(
outcome.run_dir,
selections,
performance_profile=st.session_state.get(
"performance_profile", DEFAULT_PERFORMANCE_PROFILE
),
)
except (OSError, RuntimeError, ValueError) as exc: except (OSError, RuntimeError, ValueError) as exc:
st.error(f"Protocol regeneration failed: {exc}") st.error(f"Protocol regeneration failed: {exc}")
else: else:
@@ -551,6 +558,13 @@ def main() -> None:
participants = _render_participants() participants = _render_participants()
st.header("Processing options") st.header("Processing options")
performance_profile = st.selectbox(
"Performance profile",
options=PERFORMANCE_PROFILES,
format_func=str.title,
key="performance_profile",
help="Controls protocol-generation performance using an abstract runtime profile.",
)
audio_normalization = st.checkbox( audio_normalization = st.checkbox(
"Audio normalization", "Audio normalization",
value=True, value=True,
@@ -619,6 +633,7 @@ def main() -> None:
ProcessingOptions( ProcessingOptions(
diarization_enabled=diarization_enabled, diarization_enabled=diarization_enabled,
audio_normalization=audio_normalization, audio_normalization=audio_normalization,
performance_profile=performance_profile,
), ),
progress_sink=callback, progress_sink=callback,
) )
+47
View File
@@ -561,3 +561,50 @@ def test_uploaded_source_is_preserved_in_meeting_directory(tmp_path: Path) -> No
assert destination.parent == tmp_path / "meetings" / "meeting-1" / "uploads" assert destination.parent == tmp_path / "meetings" / "meeting-1" / "uploads"
assert destination.name.endswith("_unsafe.wav") assert destination.name.endswith("_unsafe.wav")
assert destination.read_bytes() == b"source audio" assert destination.read_bytes() == b"source audio"
@pytest.mark.parametrize(
("profile", "threads"),
[("fast", 16), ("efficient", 10), ("powersave", 4)],
)
def test_process_resolves_profile_for_protocol_generation(
tmp_path: Path, profile: str, threads: int
) -> None:
service, gateway = make_service(tmp_path)
audio = tmp_path / "meeting.wav"
audio.write_bytes(b"audio")
service.process(
audio,
meeting(),
participants(),
ProcessingOptions(performance_profile=profile),
)
assert gateway.config_values is not None
assert gateway.config_values["protocol_num_thread"] == threads
def test_process_defaults_to_backend_thread_selection(tmp_path: Path) -> None:
service, gateway = make_service(tmp_path)
audio = tmp_path / "meeting.wav"
audio.write_bytes(b"audio")
service.process(audio, meeting(), participants(), ProcessingOptions())
assert gateway.config_values is not None
assert gateway.config_values["protocol_num_thread"] is None
def test_protocol_regeneration_propagates_explicit_profile(tmp_path: Path) -> None:
service, gateway = make_service(tmp_path)
write_speaker_review_artifacts(gateway.run_dir)
service.regenerate_protocol(
gateway.run_dir,
{},
performance_profile="fast",
)
assert gateway.regeneration is not None
assert gateway.regeneration["options"]["protocol_num_thread"] == 16
+21
View File
@@ -0,0 +1,21 @@
import pytest
from mka.application.meeting_service import ProcessingOptions
from mka.application.performance import PERFORMANCE_PROFILES, resolve_ollama_num_thread
def test_default_profile_is_auto() -> None:
assert ProcessingOptions().performance_profile == "auto"
assert PERFORMANCE_PROFILES[0] == "auto"
def test_auto_profile_has_no_ollama_thread_override() -> None:
assert resolve_ollama_num_thread("auto") is None
@pytest.mark.parametrize(
("profile", "threads"),
[("fast", 16), ("efficient", 10), ("powersave", 4)],
)
def test_profiles_resolve_to_current_ollama_thread_counts(profile: str, threads: int) -> None:
assert resolve_ollama_num_thread(profile) == threads