Add protocol performance profiles
This commit is contained in:
@@ -4,6 +4,9 @@
|
||||
|
||||
### Added
|
||||
|
||||
- Auto/Fast/Efficient/Powersave protocol profiles with backend-selected threads
|
||||
by default, also applied during mapped-speaker regeneration.
|
||||
|
||||
- Versioned UTF-8 YAML import/export for the SQLite terminology glossary, with
|
||||
stable entry IDs, complete validation, and atomic replacement semantics.
|
||||
- First Streamlit MVP for audio upload, Meeting Context entry, participant
|
||||
|
||||
@@ -95,6 +95,14 @@ PyTorch/ROCm processed the same duration in approximately 98.6 seconds. These
|
||||
are validation observations, not performance guarantees or hardware
|
||||
requirements. CPU execution remains supported and may be substantially slower.
|
||||
|
||||
Protocol-generation performance is selected through intentionally abstract
|
||||
profiles rather than hardware controls in the GUI. `auto` is the default and
|
||||
delegates thread selection to the inference backend; with Ollama this means
|
||||
omitting `num_thread` entirely. The explicit resource profiles currently map
|
||||
`fast` to 16 Ollama CPU threads, `efficient` to 10, and `powersave` to 4. These
|
||||
concrete mappings belong to the backend/configuration boundary and may evolve
|
||||
independently of the user-facing profile semantics.
|
||||
|
||||
## Meeting Context and Speakers
|
||||
|
||||
`MeetingContext` is a structured domain object containing meeting metadata,
|
||||
|
||||
@@ -82,6 +82,14 @@ Audio normalization is enabled by default and can be disabled in the processing
|
||||
options. This controls loudness normalization only: Meeting Lab still prepares
|
||||
every WAV, FLAC or M4A source as canonical audio before transcription.
|
||||
|
||||
The **Performance profile** selector controls protocol-generation runtime using
|
||||
the abstract Auto (default), Fast, Efficient, and Powersave profiles. Auto lets
|
||||
the inference backend select its own thread configuration; for Ollama, Meeting
|
||||
Assistant intentionally sends no `num_thread` option. Fast, Efficient, and
|
||||
Powersave are explicit resource profiles currently mapped to 16, 10, and 4
|
||||
Ollama CPU threads. These concrete mappings may evolve independently of the UI
|
||||
semantics.
|
||||
|
||||
The People section can export its current entries to a UTF-8 `people.yaml` file
|
||||
and replace them from a previous `.yaml` or `.yml` export. Stable person IDs,
|
||||
names, roles, organizations and attendance states are retained. This is a small
|
||||
|
||||
@@ -16,6 +16,7 @@ import yaml
|
||||
|
||||
from mka.application.config import AppSettings
|
||||
from mka.application.glossary import GlossaryRepository, render_glossary_terms
|
||||
from mka.application.performance import DEFAULT_PERFORMANCE_PROFILE, resolve_ollama_num_thread
|
||||
|
||||
STAGES = ("preparing", "transcription", "diarization", "protocol_generation")
|
||||
|
||||
@@ -65,6 +66,7 @@ class ParticipantInput:
|
||||
class ProcessingOptions:
|
||||
diarization_enabled: bool = False
|
||||
audio_normalization: bool = True
|
||||
performance_profile: str = DEFAULT_PERFORMANCE_PROFILE
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
@@ -229,6 +231,7 @@ class MeetingProcessingService:
|
||||
"model": self.settings.protocol_model,
|
||||
"ollama_endpoint": self.settings.ollama_endpoint,
|
||||
"protocol_num_ctx": self.settings.protocol_num_ctx,
|
||||
"protocol_num_thread": resolve_ollama_num_thread(options.performance_profile),
|
||||
"protocol_safe_input_token_budget": (
|
||||
self.settings.protocol_safe_input_token_budget
|
||||
),
|
||||
@@ -357,6 +360,7 @@ class MeetingProcessingService:
|
||||
run_dir: Path,
|
||||
speaker_mappings: dict[str, str],
|
||||
progress_sink: Callable[[AppProgressEvent], None] | None = None,
|
||||
performance_profile: str = DEFAULT_PERFORMANCE_PROFILE,
|
||||
) -> ProcessingOutcome:
|
||||
"""Persist confirmed mappings and regenerate only the direct protocol."""
|
||||
review = self.load_speaker_mapping_review(run_dir)
|
||||
@@ -401,6 +405,7 @@ class MeetingProcessingService:
|
||||
model=self.settings.protocol_model,
|
||||
ollama_endpoint=self.settings.ollama_endpoint,
|
||||
protocol_num_ctx=self.settings.protocol_num_ctx,
|
||||
protocol_num_thread=resolve_ollama_num_thread(performance_profile),
|
||||
protocol_safe_input_token_budget=(self.settings.protocol_safe_input_token_budget),
|
||||
)
|
||||
protocol_path = Path(result.protocol_path)
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
"""Abstract performance profiles resolved to current backend runtime options."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
DEFAULT_PERFORMANCE_PROFILE = "auto"
|
||||
PERFORMANCE_PROFILES = ("auto", "fast", "efficient", "powersave")
|
||||
|
||||
_OLLAMA_THREADS_BY_PROFILE = {
|
||||
"fast": 16,
|
||||
"efficient": 10,
|
||||
"powersave": 4,
|
||||
}
|
||||
|
||||
|
||||
def resolve_ollama_num_thread(profile: str) -> int | None:
|
||||
"""Resolve a profile, leaving automatic thread selection to Ollama for Auto."""
|
||||
if profile == "auto":
|
||||
return None
|
||||
try:
|
||||
return _OLLAMA_THREADS_BY_PROFILE[profile]
|
||||
except KeyError as exc:
|
||||
allowed = ", ".join(PERFORMANCE_PROFILES)
|
||||
raise ValueError(f"Unknown performance profile {profile!r}; expected: {allowed}.") from exc
|
||||
@@ -34,6 +34,7 @@ from mka.application.people_yaml import (
|
||||
export_people_yaml,
|
||||
import_people_yaml,
|
||||
)
|
||||
from mka.application.performance import DEFAULT_PERFORMANCE_PROFILE, PERFORMANCE_PROFILES
|
||||
from mka.application.run_inputs import (
|
||||
RunInputJsonError,
|
||||
RunInputState,
|
||||
@@ -495,7 +496,13 @@ def _render_result() -> None:
|
||||
):
|
||||
try:
|
||||
with st.spinner("Regenerating protocol without rerunning audio processing..."):
|
||||
regenerated = service.regenerate_protocol(outcome.run_dir, selections)
|
||||
regenerated = service.regenerate_protocol(
|
||||
outcome.run_dir,
|
||||
selections,
|
||||
performance_profile=st.session_state.get(
|
||||
"performance_profile", DEFAULT_PERFORMANCE_PROFILE
|
||||
),
|
||||
)
|
||||
except (OSError, RuntimeError, ValueError) as exc:
|
||||
st.error(f"Protocol regeneration failed: {exc}")
|
||||
else:
|
||||
@@ -551,6 +558,13 @@ def main() -> None:
|
||||
participants = _render_participants()
|
||||
|
||||
st.header("Processing options")
|
||||
performance_profile = st.selectbox(
|
||||
"Performance profile",
|
||||
options=PERFORMANCE_PROFILES,
|
||||
format_func=str.title,
|
||||
key="performance_profile",
|
||||
help="Controls protocol-generation performance using an abstract runtime profile.",
|
||||
)
|
||||
audio_normalization = st.checkbox(
|
||||
"Audio normalization",
|
||||
value=True,
|
||||
@@ -619,6 +633,7 @@ def main() -> None:
|
||||
ProcessingOptions(
|
||||
diarization_enabled=diarization_enabled,
|
||||
audio_normalization=audio_normalization,
|
||||
performance_profile=performance_profile,
|
||||
),
|
||||
progress_sink=callback,
|
||||
)
|
||||
|
||||
@@ -561,3 +561,50 @@ def test_uploaded_source_is_preserved_in_meeting_directory(tmp_path: Path) -> No
|
||||
assert destination.parent == tmp_path / "meetings" / "meeting-1" / "uploads"
|
||||
assert destination.name.endswith("_unsafe.wav")
|
||||
assert destination.read_bytes() == b"source audio"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("profile", "threads"),
|
||||
[("fast", 16), ("efficient", 10), ("powersave", 4)],
|
||||
)
|
||||
def test_process_resolves_profile_for_protocol_generation(
|
||||
tmp_path: Path, profile: str, threads: int
|
||||
) -> None:
|
||||
service, gateway = make_service(tmp_path)
|
||||
audio = tmp_path / "meeting.wav"
|
||||
audio.write_bytes(b"audio")
|
||||
|
||||
service.process(
|
||||
audio,
|
||||
meeting(),
|
||||
participants(),
|
||||
ProcessingOptions(performance_profile=profile),
|
||||
)
|
||||
|
||||
assert gateway.config_values is not None
|
||||
assert gateway.config_values["protocol_num_thread"] == threads
|
||||
|
||||
|
||||
def test_process_defaults_to_backend_thread_selection(tmp_path: Path) -> None:
|
||||
service, gateway = make_service(tmp_path)
|
||||
audio = tmp_path / "meeting.wav"
|
||||
audio.write_bytes(b"audio")
|
||||
|
||||
service.process(audio, meeting(), participants(), ProcessingOptions())
|
||||
|
||||
assert gateway.config_values is not None
|
||||
assert gateway.config_values["protocol_num_thread"] is None
|
||||
|
||||
|
||||
def test_protocol_regeneration_propagates_explicit_profile(tmp_path: Path) -> None:
|
||||
service, gateway = make_service(tmp_path)
|
||||
write_speaker_review_artifacts(gateway.run_dir)
|
||||
|
||||
service.regenerate_protocol(
|
||||
gateway.run_dir,
|
||||
{},
|
||||
performance_profile="fast",
|
||||
)
|
||||
|
||||
assert gateway.regeneration is not None
|
||||
assert gateway.regeneration["options"]["protocol_num_thread"] == 16
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
import pytest
|
||||
|
||||
from mka.application.meeting_service import ProcessingOptions
|
||||
from mka.application.performance import PERFORMANCE_PROFILES, resolve_ollama_num_thread
|
||||
|
||||
|
||||
def test_default_profile_is_auto() -> None:
|
||||
assert ProcessingOptions().performance_profile == "auto"
|
||||
assert PERFORMANCE_PROFILES[0] == "auto"
|
||||
|
||||
|
||||
def test_auto_profile_has_no_ollama_thread_override() -> None:
|
||||
assert resolve_ollama_num_thread("auto") is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("profile", "threads"),
|
||||
[("fast", 16), ("efficient", 10), ("powersave", 4)],
|
||||
)
|
||||
def test_profiles_resolve_to_current_ollama_thread_counts(profile: str, threads: int) -> None:
|
||||
assert resolve_ollama_num_thread(profile) == threads
|
||||
Reference in New Issue
Block a user