Add protocol performance profiles

This commit is contained in:
2026-09-12 11:50:30 +02:00
parent e2a0434941
commit 77be2b39ff
8 changed files with 131 additions and 1 deletions
+3
View File
@@ -4,6 +4,9 @@
### Added
- Auto/Fast/Efficient/Powersave protocol profiles with backend-selected threads
by default, also applied during mapped-speaker regeneration.
- Versioned UTF-8 YAML import/export for the SQLite terminology glossary, with
stable entry IDs, complete validation, and atomic replacement semantics.
- First Streamlit MVP for audio upload, Meeting Context entry, participant
+8
View File
@@ -95,6 +95,14 @@ PyTorch/ROCm processed the same duration in approximately 98.6 seconds. These
are validation observations, not performance guarantees or hardware
requirements. CPU execution remains supported and may be substantially slower.
Protocol-generation performance is selected through intentionally abstract
profiles rather than hardware controls in the GUI. `auto` is the default and
delegates thread selection to the inference backend; with Ollama this means
omitting `num_thread` entirely. The explicit resource profiles currently map
`fast` to 16 Ollama CPU threads, `efficient` to 10, and `powersave` to 4. These
concrete mappings belong to the backend/configuration boundary and may evolve
independently of the user-facing profile semantics.
## Meeting Context and Speakers
`MeetingContext` is a structured domain object containing meeting metadata,
+8
View File
@@ -82,6 +82,14 @@ Audio normalization is enabled by default and can be disabled in the processing
options. This controls loudness normalization only: Meeting Lab still prepares
every WAV, FLAC or M4A source as canonical audio before transcription.
The **Performance profile** selector controls protocol-generation runtime using
the abstract Auto (default), Fast, Efficient, and Powersave profiles. Auto lets
the inference backend select its own thread configuration; for Ollama, Meeting
Assistant intentionally sends no `num_thread` option. Fast, Efficient, and
Powersave are explicit resource profiles currently mapped to 16, 10, and 4
Ollama CPU threads. These concrete mappings may evolve independently of the UI
semantics.
The People section can export its current entries to a UTF-8 `people.yaml` file
and replace them from a previous `.yaml` or `.yml` export. Stable person IDs,
names, roles, organizations and attendance states are retained. This is a small
+5
View File
@@ -16,6 +16,7 @@ import yaml
from mka.application.config import AppSettings
from mka.application.glossary import GlossaryRepository, render_glossary_terms
from mka.application.performance import DEFAULT_PERFORMANCE_PROFILE, resolve_ollama_num_thread
STAGES = ("preparing", "transcription", "diarization", "protocol_generation")
@@ -65,6 +66,7 @@ class ParticipantInput:
class ProcessingOptions:
diarization_enabled: bool = False
audio_normalization: bool = True
performance_profile: str = DEFAULT_PERFORMANCE_PROFILE
@dataclass(frozen=True)
@@ -229,6 +231,7 @@ class MeetingProcessingService:
"model": self.settings.protocol_model,
"ollama_endpoint": self.settings.ollama_endpoint,
"protocol_num_ctx": self.settings.protocol_num_ctx,
"protocol_num_thread": resolve_ollama_num_thread(options.performance_profile),
"protocol_safe_input_token_budget": (
self.settings.protocol_safe_input_token_budget
),
@@ -357,6 +360,7 @@ class MeetingProcessingService:
run_dir: Path,
speaker_mappings: dict[str, str],
progress_sink: Callable[[AppProgressEvent], None] | None = None,
performance_profile: str = DEFAULT_PERFORMANCE_PROFILE,
) -> ProcessingOutcome:
"""Persist confirmed mappings and regenerate only the direct protocol."""
review = self.load_speaker_mapping_review(run_dir)
@@ -401,6 +405,7 @@ class MeetingProcessingService:
model=self.settings.protocol_model,
ollama_endpoint=self.settings.ollama_endpoint,
protocol_num_ctx=self.settings.protocol_num_ctx,
protocol_num_thread=resolve_ollama_num_thread(performance_profile),
protocol_safe_input_token_budget=(self.settings.protocol_safe_input_token_budget),
)
protocol_path = Path(result.protocol_path)
+23
View File
@@ -0,0 +1,23 @@
"""Abstract performance profiles resolved to current backend runtime options."""
from __future__ import annotations
DEFAULT_PERFORMANCE_PROFILE = "auto"
PERFORMANCE_PROFILES = ("auto", "fast", "efficient", "powersave")
_OLLAMA_THREADS_BY_PROFILE = {
"fast": 16,
"efficient": 10,
"powersave": 4,
}
def resolve_ollama_num_thread(profile: str) -> int | None:
"""Resolve a profile, leaving automatic thread selection to Ollama for Auto."""
if profile == "auto":
return None
try:
return _OLLAMA_THREADS_BY_PROFILE[profile]
except KeyError as exc:
allowed = ", ".join(PERFORMANCE_PROFILES)
raise ValueError(f"Unknown performance profile {profile!r}; expected: {allowed}.") from exc
+16 -1
View File
@@ -34,6 +34,7 @@ from mka.application.people_yaml import (
export_people_yaml,
import_people_yaml,
)
from mka.application.performance import DEFAULT_PERFORMANCE_PROFILE, PERFORMANCE_PROFILES
from mka.application.run_inputs import (
RunInputJsonError,
RunInputState,
@@ -495,7 +496,13 @@ def _render_result() -> None:
):
try:
with st.spinner("Regenerating protocol without rerunning audio processing..."):
regenerated = service.regenerate_protocol(outcome.run_dir, selections)
regenerated = service.regenerate_protocol(
outcome.run_dir,
selections,
performance_profile=st.session_state.get(
"performance_profile", DEFAULT_PERFORMANCE_PROFILE
),
)
except (OSError, RuntimeError, ValueError) as exc:
st.error(f"Protocol regeneration failed: {exc}")
else:
@@ -551,6 +558,13 @@ def main() -> None:
participants = _render_participants()
st.header("Processing options")
performance_profile = st.selectbox(
"Performance profile",
options=PERFORMANCE_PROFILES,
format_func=str.title,
key="performance_profile",
help="Controls protocol-generation performance using an abstract runtime profile.",
)
audio_normalization = st.checkbox(
"Audio normalization",
value=True,
@@ -619,6 +633,7 @@ def main() -> None:
ProcessingOptions(
diarization_enabled=diarization_enabled,
audio_normalization=audio_normalization,
performance_profile=performance_profile,
),
progress_sink=callback,
)
+47
View File
@@ -561,3 +561,50 @@ def test_uploaded_source_is_preserved_in_meeting_directory(tmp_path: Path) -> No
assert destination.parent == tmp_path / "meetings" / "meeting-1" / "uploads"
assert destination.name.endswith("_unsafe.wav")
assert destination.read_bytes() == b"source audio"
@pytest.mark.parametrize(
("profile", "threads"),
[("fast", 16), ("efficient", 10), ("powersave", 4)],
)
def test_process_resolves_profile_for_protocol_generation(
tmp_path: Path, profile: str, threads: int
) -> None:
service, gateway = make_service(tmp_path)
audio = tmp_path / "meeting.wav"
audio.write_bytes(b"audio")
service.process(
audio,
meeting(),
participants(),
ProcessingOptions(performance_profile=profile),
)
assert gateway.config_values is not None
assert gateway.config_values["protocol_num_thread"] == threads
def test_process_defaults_to_backend_thread_selection(tmp_path: Path) -> None:
service, gateway = make_service(tmp_path)
audio = tmp_path / "meeting.wav"
audio.write_bytes(b"audio")
service.process(audio, meeting(), participants(), ProcessingOptions())
assert gateway.config_values is not None
assert gateway.config_values["protocol_num_thread"] is None
def test_protocol_regeneration_propagates_explicit_profile(tmp_path: Path) -> None:
service, gateway = make_service(tmp_path)
write_speaker_review_artifacts(gateway.run_dir)
service.regenerate_protocol(
gateway.run_dir,
{},
performance_profile="fast",
)
assert gateway.regeneration is not None
assert gateway.regeneration["options"]["protocol_num_thread"] == 16
+21
View File
@@ -0,0 +1,21 @@
import pytest
from mka.application.meeting_service import ProcessingOptions
from mka.application.performance import PERFORMANCE_PROFILES, resolve_ollama_num_thread
def test_default_profile_is_auto() -> None:
assert ProcessingOptions().performance_profile == "auto"
assert PERFORMANCE_PROFILES[0] == "auto"
def test_auto_profile_has_no_ollama_thread_override() -> None:
assert resolve_ollama_num_thread("auto") is None
@pytest.mark.parametrize(
("profile", "threads"),
[("fast", 16), ("efficient", 10), ("powersave", 4)],
)
def test_profiles_resolve_to_current_ollama_thread_counts(profile: str, threads: int) -> None:
assert resolve_ollama_num_thread(profile) == threads