735 lines
25 KiB
Python
735 lines
25 KiB
Python
import json
|
|
from dataclasses import dataclass, replace
|
|
from datetime import date
|
|
from pathlib import Path
|
|
from types import SimpleNamespace
|
|
from typing import Any
|
|
|
|
import pytest
|
|
|
|
from mka.application.config import AppSettings
|
|
from mka.application.meeting_service import (
|
|
MeetingDetails,
|
|
MeetingProcessingService,
|
|
ParticipantInput,
|
|
ProcessingOptions,
|
|
)
|
|
from mka.application.progress_timing import ProcessingTimer
|
|
|
|
|
|
@dataclass
|
|
class FakeContext:
|
|
data: dict[str, Any]
|
|
|
|
@property
|
|
def meeting_id(self) -> str:
|
|
return self.data["meeting"]["meeting_id"]
|
|
|
|
|
|
class FakeMeetingLab:
|
|
def __init__(self, run_dir: Path) -> None:
|
|
self.run_dir = run_dir
|
|
self.context_data: dict[str, Any] | None = None
|
|
self.config_values: dict[str, Any] | None = None
|
|
self.fail = False
|
|
self.regeneration: dict[str, Any] | None = None
|
|
|
|
def create_context(self, data: dict[str, Any]) -> FakeContext:
|
|
self.context_data = data
|
|
if not data["meeting"]["title"]:
|
|
raise ValueError("meeting.title must be present and non-empty")
|
|
participant_ids = [item["participant_id"] for item in data["participants"]]
|
|
if len(participant_ids) != len(set(participant_ids)):
|
|
raise ValueError("Duplicate participant_id")
|
|
return FakeContext(data)
|
|
|
|
def create_config(self, values: dict[str, Any]) -> dict[str, Any]:
|
|
self.config_values = values
|
|
return values
|
|
|
|
def run(self, config: Any, meeting_context: Any, progress_sink: Any) -> Any:
|
|
self.run_dir.mkdir(parents=True, exist_ok=True)
|
|
progress_sink(
|
|
SimpleNamespace(
|
|
stage="preparing",
|
|
status="started",
|
|
elapsed_seconds=0.1,
|
|
progress=None,
|
|
message=None,
|
|
)
|
|
)
|
|
progress_sink(
|
|
SimpleNamespace(
|
|
stage="preparing",
|
|
status="completed",
|
|
elapsed_seconds=0.2,
|
|
progress=None,
|
|
message=None,
|
|
)
|
|
)
|
|
progress_sink(
|
|
SimpleNamespace(
|
|
stage="transcription",
|
|
status="started",
|
|
elapsed_seconds=0.2,
|
|
progress=0.25,
|
|
message="transcribing",
|
|
)
|
|
)
|
|
if self.fail:
|
|
(self.run_dir / "run_metadata.json").write_text(
|
|
json.dumps(
|
|
{
|
|
"failure": {
|
|
"stage": "whisper",
|
|
"type": "TranscriptionError",
|
|
"message": "model failed",
|
|
}
|
|
}
|
|
),
|
|
encoding="utf-8",
|
|
)
|
|
progress_sink(
|
|
SimpleNamespace(
|
|
stage="failed",
|
|
status="failed",
|
|
elapsed_seconds=0.3,
|
|
progress=None,
|
|
message="transcription: model failed",
|
|
)
|
|
)
|
|
return SimpleNamespace(exit_code=2, run_dir=self.run_dir, protocol_path=None)
|
|
protocol = self.run_dir / "protocol.md"
|
|
protocol.write_text("# Generated protocol\n", encoding="utf-8")
|
|
return SimpleNamespace(exit_code=0, run_dir=self.run_dir, protocol_path=protocol)
|
|
|
|
def regenerate_protocol(
|
|
self,
|
|
run_dir: Path,
|
|
meeting_context: Any,
|
|
progress_sink: Any,
|
|
**options: Any,
|
|
) -> Any:
|
|
self.regeneration = {
|
|
"run_dir": run_dir,
|
|
"meeting_context": meeting_context,
|
|
"options": options,
|
|
}
|
|
progress_sink(
|
|
SimpleNamespace(
|
|
stage="protocol_generation",
|
|
status="started",
|
|
elapsed_seconds=0.0,
|
|
progress=None,
|
|
message=None,
|
|
)
|
|
)
|
|
protocol = Path(run_dir) / "protocol.md"
|
|
protocol.write_text("# Regenerated protocol\n", encoding="utf-8")
|
|
protocol_dir = Path(run_dir) / "protocol"
|
|
protocol_dir.mkdir(exist_ok=True)
|
|
(protocol_dir / "runtime_metadata.json").write_text(
|
|
json.dumps({"speaker_attribution_available": True}), encoding="utf-8"
|
|
)
|
|
return SimpleNamespace(exit_code=0, run_dir=run_dir, protocol_path=protocol)
|
|
|
|
|
|
def make_service(tmp_path: Path) -> tuple[MeetingProcessingService, FakeMeetingLab]:
|
|
model = tmp_path / "model.bin"
|
|
model.write_bytes(b"model")
|
|
gateway = FakeMeetingLab(tmp_path / "backend-run")
|
|
settings = AppSettings(
|
|
data_root=tmp_path / "meetings",
|
|
whisper_model=model,
|
|
glossary_database=tmp_path / "glossary.sqlite3",
|
|
whisper_executable="/opt/whisper-cli",
|
|
protocol_model="test:model",
|
|
diarization_mode="gpu",
|
|
)
|
|
return MeetingProcessingService(settings, gateway), gateway
|
|
|
|
|
|
def meeting() -> MeetingDetails:
|
|
return MeetingDetails(
|
|
title="Architecture Review",
|
|
language="de",
|
|
meeting_date=date(2026, 8, 23),
|
|
description="Review the MVP.",
|
|
)
|
|
|
|
|
|
def participants() -> list[ParticipantInput]:
|
|
return [
|
|
ParticipantInput(
|
|
participant_id="martin",
|
|
display_name="Martin",
|
|
role="Project lead",
|
|
organization="Engineering",
|
|
),
|
|
ParticipantInput(
|
|
participant_id="alex",
|
|
display_name="Alex",
|
|
organization="Engineering",
|
|
),
|
|
]
|
|
|
|
|
|
def write_speaker_review_artifacts(run_dir: Path) -> Path:
|
|
diarization_dir = run_dir / "diarization"
|
|
context_dir = run_dir / "context"
|
|
diarization_dir.mkdir(parents=True)
|
|
context_dir.mkdir()
|
|
transcript_path = diarization_dir / "transcript_diarized.json"
|
|
transcript_path.write_text(
|
|
json.dumps(
|
|
{
|
|
"speaker_labels_anonymous": True,
|
|
"segments": [
|
|
{
|
|
"speaker_id": "SPEAKER_01",
|
|
"text": "I will prepare all raw materials before Wednesday.",
|
|
},
|
|
{"speaker_id": "SPEAKER_00", "text": "Yes."},
|
|
{
|
|
"speaker_id": "SPEAKER_00",
|
|
"text": "We will run the production trial on Wednesday.",
|
|
},
|
|
{
|
|
"speaker_id": "SPEAKER_00",
|
|
"text": "The trial requires the complete production team.",
|
|
},
|
|
{"speaker_id": None, "text": "Unassigned text."},
|
|
],
|
|
}
|
|
),
|
|
encoding="utf-8",
|
|
)
|
|
(context_dir / "meeting_context.yaml").write_text(
|
|
json.dumps(
|
|
{
|
|
"schema_version": "1",
|
|
"meeting": {
|
|
"meeting_id": "speaker-review",
|
|
"title": "Speaker review",
|
|
"language": "en",
|
|
},
|
|
"participants": [
|
|
{"participant_id": "martin", "display_name": "Martin"},
|
|
{"participant_id": "anna", "display_name": "Anna"},
|
|
],
|
|
"speaker_mappings": {},
|
|
"mentioned_people": [],
|
|
"organization": {"departments": []},
|
|
"known_entities": {},
|
|
}
|
|
),
|
|
encoding="utf-8",
|
|
)
|
|
return transcript_path
|
|
|
|
|
|
def test_build_context_uses_actual_v1_shape(tmp_path: Path) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
|
|
context = service.build_context(meeting(), participants())
|
|
|
|
assert context.meeting_id == "architecture-review"
|
|
assert gateway.context_data is not None
|
|
assert gateway.context_data["meeting"]["date"] == "2026-08-23"
|
|
assert gateway.context_data["meeting"]["notes"] == "Review the MVP."
|
|
assert gateway.context_data["participants"][0] == {
|
|
"participant_id": "martin",
|
|
"display_name": "Martin",
|
|
"aliases": [],
|
|
"role": "Project lead",
|
|
"department": "engineering",
|
|
"attendance_status": "present",
|
|
"notes": None,
|
|
}
|
|
assert gateway.context_data["organization"]["departments"] == [
|
|
{"id": "engineering", "name": "Engineering", "aliases": []}
|
|
]
|
|
|
|
|
|
def test_build_context_preserves_explicit_speaker_mapping(tmp_path: Path) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
|
|
service.build_context(meeting(), participants(), {"SPEAKER_00": "martin"})
|
|
|
|
assert gateway.context_data is not None
|
|
assert gateway.context_data["speaker_mappings"] == {"SPEAKER_00": "martin"}
|
|
|
|
|
|
def test_build_context_includes_only_active_authoritative_glossary_terms(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
service.glossary.create("Secugrid HS", "product", aliases=("Sikirgut", "Secugrid H S"))
|
|
inactive = service.glossary.create("Old Name", "other")
|
|
service.glossary.set_active(inactive.id, False)
|
|
|
|
service.build_context(meeting(), participants())
|
|
|
|
assert gateway.context_data is not None
|
|
assert gateway.context_data["known_entities"] == {
|
|
"Authoritative terminology": ["Secugrid HS (aliases: Secugrid H S, Sikirgut)"]
|
|
}
|
|
rules = gateway.context_data["context_rules"]
|
|
assert "do not invent matches" in rules["glossary_canonical_spelling"]
|
|
assert "inside compounds" in rules["glossary_core_terms"]
|
|
assert "Old Name" not in str(gateway.context_data)
|
|
|
|
|
|
def test_glossary_is_rendered_into_meeting_lab_protocol_context(tmp_path: Path) -> None:
|
|
meeting_context = pytest.importorskip("src.meeting_lab.models.meeting_context")
|
|
service, gateway = make_service(tmp_path)
|
|
service.glossary.create("PBAT", "acronym", aliases=("P B A T",))
|
|
context = service.build_context(meeting(), participants())
|
|
|
|
real_context = meeting_context.create_meeting_context(context.data)
|
|
prompt_context = meeting_context.render_meeting_context_for_prompt(real_context)
|
|
|
|
assert "Authoritative terminology: PBAT (aliases: P B A T)" in prompt_context
|
|
assert "Use canonical glossary spellings" in prompt_context
|
|
assert "use canonical core terms inside compounds" in prompt_context
|
|
|
|
|
|
def test_build_context_translates_mentioned_only_person(tmp_path: Path) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
people = participants() + [
|
|
ParticipantInput(
|
|
participant_id="sam",
|
|
display_name="Sam",
|
|
attendance_status="mentioned_only",
|
|
)
|
|
]
|
|
|
|
service.build_context(meeting(), people)
|
|
|
|
assert gateway.context_data is not None
|
|
assert [item["participant_id"] for item in gateway.context_data["participants"]] == [
|
|
"martin",
|
|
"alex",
|
|
]
|
|
assert gateway.context_data["mentioned_people"] == [
|
|
{
|
|
"person_id": "sam",
|
|
"display_name": "Sam",
|
|
"aliases": [],
|
|
"role": None,
|
|
"department": None,
|
|
"attendance_status": "mentioned_only",
|
|
"notes": None,
|
|
}
|
|
]
|
|
|
|
|
|
@pytest.mark.parametrize("language", ["de", "en"])
|
|
def test_process_translates_configuration_and_disables_diarization(
|
|
tmp_path: Path,
|
|
language: str,
|
|
) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
audio = tmp_path / "meeting.wav"
|
|
audio.write_bytes(b"audio")
|
|
|
|
outcome = service.process(
|
|
audio,
|
|
replace(meeting(), language=language),
|
|
participants(),
|
|
ProcessingOptions(diarization_enabled=False),
|
|
)
|
|
|
|
assert outcome.succeeded
|
|
assert gateway.config_values is not None
|
|
assert gateway.config_values["language"] == language
|
|
assert service.build_context(replace(meeting(), language=language), participants()).data[
|
|
"meeting"
|
|
]["language"] == language
|
|
assert gateway.config_values["diarization"] == "off"
|
|
assert gateway.config_values["model"] == "test:model"
|
|
assert gateway.config_values["protocol_num_ctx"] == 32_768
|
|
assert gateway.config_values["protocol_safe_input_token_budget"] == 29_000
|
|
assert gateway.config_values["whisper_executable"] == "/opt/whisper-cli"
|
|
assert gateway.config_values["ffmpeg_executable"] == "ffmpeg"
|
|
assert gateway.config_values["audio_normalization"] is True
|
|
assert gateway.config_values["diarization_runtime"] == "native"
|
|
assert gateway.config_values["diarization_container_args"] == ()
|
|
assert gateway.config_values["output_root"] == (
|
|
tmp_path / "meetings" / "architecture-review" / "runs"
|
|
)
|
|
|
|
|
|
def test_process_propagates_disabled_audio_normalization(tmp_path: Path) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
audio = tmp_path / "meeting.m4a"
|
|
audio.write_bytes(b"audio")
|
|
|
|
outcome = service.process(
|
|
audio,
|
|
meeting(),
|
|
participants(),
|
|
ProcessingOptions(audio_normalization=False),
|
|
)
|
|
|
|
assert outcome.succeeded
|
|
assert gateway.config_values is not None
|
|
assert gateway.config_values["audio_normalization"] is False
|
|
|
|
|
|
def test_processing_options_default_to_audio_normalization_on() -> None:
|
|
assert ProcessingOptions().audio_normalization is True
|
|
|
|
|
|
def test_process_propagates_enabled_diarization_and_progress(tmp_path: Path) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
audio = tmp_path / "meeting.flac"
|
|
audio.write_bytes(b"audio")
|
|
events = []
|
|
|
|
service.process(
|
|
audio,
|
|
meeting(),
|
|
participants(),
|
|
ProcessingOptions(diarization_enabled=True),
|
|
progress_sink=events.append,
|
|
)
|
|
|
|
assert gateway.config_values is not None
|
|
assert gateway.config_values["diarization"] == "gpu"
|
|
assert [(event.stage, event.status) for event in events[:2]] == [
|
|
("preparing", "started"),
|
|
("preparing", "completed"),
|
|
]
|
|
assert events[2].progress == 0.25
|
|
|
|
|
|
def test_process_propagates_ordered_diarization_container_args(tmp_path: Path) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
service.settings = replace(
|
|
service.settings,
|
|
diarization_runtime="container",
|
|
diarization_container_image="runtime/image:tag",
|
|
diarization_container_args=(
|
|
"--network=host",
|
|
"--label",
|
|
"meeting-test",
|
|
),
|
|
)
|
|
audio = tmp_path / "meeting.wav"
|
|
audio.write_bytes(b"audio")
|
|
|
|
outcome = service.process(
|
|
audio,
|
|
meeting(),
|
|
participants(),
|
|
ProcessingOptions(diarization_enabled=True),
|
|
)
|
|
|
|
assert outcome.succeeded
|
|
assert gateway.config_values is not None
|
|
assert gateway.config_values["diarization_container_args"] == (
|
|
"--network=host",
|
|
"--label",
|
|
"meeting-test",
|
|
)
|
|
|
|
|
|
def test_result_and_user_edit_are_preserved_separately(tmp_path: Path) -> None:
|
|
service, _ = make_service(tmp_path)
|
|
audio = tmp_path / "meeting.wav"
|
|
audio.write_bytes(b"audio")
|
|
|
|
outcome = service.process(audio, meeting(), participants(), ProcessingOptions())
|
|
edited_path = service.save_edited_protocol(outcome.run_dir, "# Reviewed\n")
|
|
|
|
assert outcome.original_protocol == "# Generated protocol\n"
|
|
assert outcome.protocol_path.read_text(encoding="utf-8") == "# Generated protocol\n"
|
|
assert edited_path.read_text(encoding="utf-8") == "# Reviewed\n"
|
|
|
|
|
|
def test_failure_reports_stage_and_preserves_run_dir(tmp_path: Path) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
gateway.fail = True
|
|
audio = tmp_path / "meeting.wav"
|
|
audio.write_bytes(b"audio")
|
|
|
|
outcome = service.process(audio, meeting(), participants(), ProcessingOptions())
|
|
|
|
assert not outcome.succeeded
|
|
assert outcome.failed_stage == "transcription"
|
|
assert outcome.error_message == "TranscriptionError: model failed"
|
|
assert outcome.run_dir == gateway.run_dir
|
|
assert (gateway.run_dir / "run_metadata.json").is_file()
|
|
|
|
|
|
def test_speaker_review_lists_detected_labels_participants_and_excerpts(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
source = write_speaker_review_artifacts(gateway.run_dir)
|
|
|
|
review = service.load_speaker_mapping_review(gateway.run_dir, excerpts_per_speaker=2)
|
|
|
|
assert review is not None
|
|
assert [speaker.speaker_label for speaker in review.speakers] == [
|
|
"SPEAKER_00",
|
|
"SPEAKER_01",
|
|
]
|
|
assert review.speakers[0].excerpts == (
|
|
"We will run the production trial on Wednesday.",
|
|
"The trial requires the complete production team.",
|
|
)
|
|
assert review.participants == (("martin", "Martin"), ("anna", "Anna"))
|
|
assert review.current_mappings == {}
|
|
assert "SPEAKER_00" in source.read_text(encoding="utf-8")
|
|
|
|
|
|
def test_protocol_only_regeneration_persists_mappings_without_rewriting_transcript(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
service.glossary.create("ENLYZE", "organization", aliases=("Enlyse",))
|
|
source = write_speaker_review_artifacts(gateway.run_dir)
|
|
original_source = source.read_bytes()
|
|
events = []
|
|
|
|
outcome = service.regenerate_protocol(
|
|
gateway.run_dir,
|
|
{"SPEAKER_00": "martin"},
|
|
progress_sink=events.append,
|
|
)
|
|
|
|
assert outcome.succeeded
|
|
assert outcome.original_protocol == "# Regenerated protocol\n"
|
|
assert outcome.speaker_attribution_available is True
|
|
assert gateway.context_data is not None
|
|
assert gateway.context_data["speaker_mappings"] == {"SPEAKER_00": "martin"}
|
|
assert gateway.context_data["known_entities"] == {
|
|
"Authoritative terminology": ["ENLYZE (aliases: Enlyse)"]
|
|
}
|
|
assert gateway.regeneration is not None
|
|
assert gateway.regeneration["meeting_context"].data["speaker_mappings"] == {
|
|
"SPEAKER_00": "martin"
|
|
}
|
|
assert gateway.regeneration["options"]["protocol_num_ctx"] == 32_768
|
|
assert gateway.regeneration["options"]["glossary_aliases"] == {"Enlyse": "ENLYZE"}
|
|
assert events[0].stage == "protocol_generation"
|
|
assert source.read_bytes() == original_source
|
|
|
|
|
|
def test_protocol_regeneration_allows_unmapped_and_rejects_duplicate_participant(
|
|
tmp_path: Path,
|
|
) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
write_speaker_review_artifacts(gateway.run_dir)
|
|
|
|
outcome = service.regenerate_protocol(gateway.run_dir, {})
|
|
|
|
assert outcome.succeeded
|
|
assert gateway.context_data is not None
|
|
assert gateway.context_data["speaker_mappings"] == {}
|
|
|
|
with pytest.raises(ValueError, match="only one speaker"):
|
|
service.regenerate_protocol(
|
|
gateway.run_dir,
|
|
{"SPEAKER_00": "martin", "SPEAKER_01": "martin"},
|
|
)
|
|
|
|
|
|
def test_fallback_attribution_loss_is_read_from_runtime_metadata(tmp_path: Path) -> None:
|
|
run_dir = tmp_path / "run"
|
|
protocol_dir = run_dir / "protocol"
|
|
protocol_dir.mkdir(parents=True)
|
|
(protocol_dir / "runtime_metadata.json").write_text(
|
|
json.dumps(
|
|
{
|
|
"speaker_attribution_available": False,
|
|
"speaker_attribution_loss_reason": "plain_transcript_fallback",
|
|
}
|
|
),
|
|
encoding="utf-8",
|
|
)
|
|
|
|
assert MeetingProcessingService._speaker_attribution_available(run_dir) is False
|
|
|
|
|
|
def test_uploaded_source_is_preserved_in_meeting_directory(tmp_path: Path) -> None:
|
|
service, _ = make_service(tmp_path)
|
|
source = SimpleNamespace(getbuffer=lambda: b"source audio")
|
|
|
|
destination = service.preserve_upload("meeting-1", "../unsafe.wav", source)
|
|
|
|
assert destination.parent == tmp_path / "meetings" / "meeting-1" / "uploads"
|
|
assert destination.name.endswith("_unsafe.wav")
|
|
assert destination.read_bytes() == b"source audio"
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("profile", "threads"),
|
|
[("fast", 16), ("efficient", 10), ("powersave", 4)],
|
|
)
|
|
def test_process_resolves_profile_for_protocol_generation(
|
|
tmp_path: Path, profile: str, threads: int
|
|
) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
audio = tmp_path / "meeting.wav"
|
|
audio.write_bytes(b"audio")
|
|
|
|
service.process(
|
|
audio,
|
|
meeting(),
|
|
participants(),
|
|
ProcessingOptions(performance_profile=profile),
|
|
)
|
|
|
|
assert gateway.config_values is not None
|
|
assert gateway.config_values["protocol_num_thread"] == threads
|
|
|
|
|
|
def test_process_defaults_to_backend_thread_selection(tmp_path: Path) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
audio = tmp_path / "meeting.wav"
|
|
audio.write_bytes(b"audio")
|
|
|
|
service.process(audio, meeting(), participants(), ProcessingOptions())
|
|
|
|
assert gateway.config_values is not None
|
|
assert gateway.config_values["protocol_num_thread"] is None
|
|
|
|
|
|
def test_protocol_regeneration_propagates_explicit_profile(tmp_path: Path) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
write_speaker_review_artifacts(gateway.run_dir)
|
|
|
|
service.regenerate_protocol(
|
|
gateway.run_dir,
|
|
{},
|
|
performance_profile="fast",
|
|
)
|
|
|
|
assert gateway.regeneration is not None
|
|
assert gateway.regeneration["options"]["protocol_num_thread"] == 16
|
|
|
|
|
|
def test_process_passes_only_active_explicit_glossary_aliases(tmp_path: Path) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
service.glossary.create("Luminy", "product", aliases=("Lumini",))
|
|
inactive = service.glossary.create("PBAT", "acronym", aliases=("PBRT",))
|
|
service.glossary.set_active(inactive.id, False)
|
|
audio = tmp_path / "meeting.wav"
|
|
audio.write_bytes(b"audio")
|
|
|
|
service.process(audio, meeting(), participants(), ProcessingOptions())
|
|
|
|
assert gateway.config_values is not None
|
|
assert gateway.config_values["glossary_aliases"] == {"Lumini": "Luminy"}
|
|
|
|
|
|
@pytest.mark.parametrize("fails", [False, True])
|
|
def test_process_persists_completed_and_failed_timings(tmp_path: Path, fails: bool) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
gateway.fail = fails
|
|
values = iter([0.0, 1.0, 9.0, 10.0, 20.0])
|
|
service.timer = ProcessingTimer(lambda: next(values))
|
|
audio = tmp_path / "meeting.wav"
|
|
audio.write_bytes(b"audio")
|
|
|
|
outcome = service.process(audio, meeting(), participants(), ProcessingOptions())
|
|
|
|
metadata = json.loads((gateway.run_dir / "run_metadata.json").read_text(encoding="utf-8"))
|
|
assert metadata["timing"] == {
|
|
"stages_seconds": {"preparing": 8.0, "transcription": 10.0},
|
|
"total_seconds": 20.0,
|
|
}
|
|
assert outcome.succeeded is not fails
|
|
if fails:
|
|
assert metadata["failure"]["type"] == "TranscriptionError"
|
|
|
|
|
|
def test_regeneration_starts_fresh_timer_and_retains_final_duration(
|
|
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
write_speaker_review_artifacts(gateway.run_dir)
|
|
clock = SimpleNamespace(value=10.0)
|
|
service.timer = ProcessingTimer(lambda: clock.value)
|
|
audio = tmp_path / "meeting.wav"
|
|
audio.write_bytes(b"audio")
|
|
service.process(audio, meeting(), participants(), ProcessingOptions())
|
|
|
|
clock.value = 100.0
|
|
observed = []
|
|
|
|
def regenerate(run_dir: Path, context: Any, progress_sink: Any, **options: Any) -> Any:
|
|
progress_sink(
|
|
SimpleNamespace(
|
|
stage="protocol_generation",
|
|
status="started",
|
|
elapsed_seconds=0.0,
|
|
progress=None,
|
|
message=None,
|
|
)
|
|
)
|
|
clock.value = 105.0
|
|
observed.append(service.timer.snapshot())
|
|
progress_sink(
|
|
SimpleNamespace(
|
|
stage="protocol_generation",
|
|
status="completed",
|
|
elapsed_seconds=5.0,
|
|
progress=None,
|
|
message=None,
|
|
)
|
|
)
|
|
clock.value = 109.0
|
|
protocol = run_dir / "protocol.md"
|
|
protocol.write_text("# Regenerated protocol\n", encoding="utf-8")
|
|
return SimpleNamespace(run_dir=run_dir, protocol_path=protocol)
|
|
|
|
monkeypatch.setattr(gateway, "regenerate_protocol", regenerate)
|
|
|
|
service.regenerate_protocol(gateway.run_dir, {"SPEAKER_00": "martin"})
|
|
final = service.timer.snapshot()
|
|
|
|
assert observed[0].active_stage == "protocol_generation"
|
|
assert observed[0].stage_durations == {"protocol_generation": 5.0}
|
|
assert observed[0].total_duration == 5.0
|
|
assert observed[0].running is True
|
|
assert final.stage_durations == {"protocol_generation": 5.0}
|
|
assert final.total_duration == 9.0
|
|
assert final.running is False
|
|
|
|
|
|
def test_failed_regeneration_freezes_active_and_total_durations(
|
|
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
|
) -> None:
|
|
service, gateway = make_service(tmp_path)
|
|
write_speaker_review_artifacts(gateway.run_dir)
|
|
clock = SimpleNamespace(value=20.0)
|
|
service.timer = ProcessingTimer(lambda: clock.value)
|
|
|
|
def fail_regeneration(run_dir: Path, context: Any, progress_sink: Any, **options: Any) -> Any:
|
|
progress_sink(
|
|
SimpleNamespace(
|
|
stage="protocol_generation",
|
|
status="started",
|
|
elapsed_seconds=0.0,
|
|
progress=None,
|
|
message=None,
|
|
)
|
|
)
|
|
clock.value = 27.0
|
|
raise RuntimeError("generation failed")
|
|
|
|
monkeypatch.setattr(gateway, "regenerate_protocol", fail_regeneration)
|
|
|
|
with pytest.raises(RuntimeError, match="generation failed"):
|
|
service.regenerate_protocol(gateway.run_dir, {})
|
|
clock.value = 99.0
|
|
final = service.timer.snapshot()
|
|
|
|
assert final.stage_durations == {"protocol_generation": 7.0}
|
|
assert final.total_duration == 7.0
|
|
assert final.running is False
|