diff --git a/PROJECT_KNOWLEDGE.md b/PROJECT_KNOWLEDGE.md index f543add..d67ce13 100644 --- a/PROJECT_KNOWLEDGE.md +++ b/PROJECT_KNOWLEDGE.md @@ -25,6 +25,13 @@ Implemented: - Prompt loading from `src/meeting_lab/llm/prompts.py`. - Meeting Context V1 loading, validation and optional extraction prompt injection with minimal extraction JSON provenance. +- FFmpeg-backed WAV, FLAC and M4A preparation into a per-run canonical mono + 16 kHz signed PCM16 WAV artifact before transcription or diarization. Audio + preparation always runs. Optional loudness normalization defaults to on and + currently uses the isolated FFmpeg filter + `loudnorm=I=-16:LRA=11:TP=-1.5`. This is a conservative speech-recording + default and may be revisited after empirical comparison without changing the + orchestration API. - Interim Markdown protocol generation in `src/meeting_lab/protocol/`. - Non-LLM unit tests for chunking, extraction helpers, protocol rendering and gold-test runner validation. @@ -139,6 +146,9 @@ departments only when they are explicitly supplied as metadata. It must not be used to infer responsibilities. In the current implementation this context can be injected into chunk extraction prompts as authoritative metadata, and only minimal provenance is written to extraction JSON. +The implemented MVP statuses are exactly `present` and `mentioned_only`. +Legacy entries without a status receive collection-appropriate defaults. Only +present participants may be targets of explicit `SPEAKER_XX` mappings. A `responsible` or future `owner` / `assignee` value may be recorded only when source evidence explicitly assigns, accepts or confirms responsibility. If the diff --git a/docs/meeting-context.md b/docs/meeting-context.md index ce62a77..6261941 100644 --- a/docs/meeting-context.md +++ b/docs/meeting-context.md @@ -107,8 +107,10 @@ or aliases are corrected. `department`: Organizational unit. Optional and nullable. -`attendance_status`: `present` for participants. This distinguishes attendees -from mentioned people. +`attendance_status`: exactly `present` for participants or `mentioned_only` for +people who are relevant but did not attend. For backward compatibility, a +missing status defaults to `present` in `participants` and `mentioned_only` in +`mentioned_people`. `mentioned_people`: People discussed or referenced but not present. They are not participants and must not be treated as speakers. @@ -153,7 +155,7 @@ mentioned_people: aliases: [] role: null department: null - attendance_status: "not_present" + attendance_status: "mentioned_only" notes: "Wurde erwaehnt, war aber nicht anwesend." organization: @@ -282,6 +284,8 @@ The current validator checks that: - participant ids and mentioned-person ids do not collide - referenced departments exist in `organization.departments` - `attendance_status` values are valid +- speaker mappings reference present participants only; mentioned-only people + cannot be diarized speakers - participants are marked `present` - mentioned people are not marked `present` diff --git a/samples/real_live/progeo_meeting/meeting_context.yaml b/samples/real_live/progeo_meeting/meeting_context.yaml index 3f2a994..521fc2a 100644 --- a/samples/real_live/progeo_meeting/meeting_context.yaml +++ b/samples/real_live/progeo_meeting/meeting_context.yaml @@ -75,7 +75,7 @@ mentioned_people: aliases: [] role: null department: null - attendance_status: "not_present" + attendance_status: "mentioned_only" notes: null organization: @@ -127,4 +127,4 @@ context_rules: do_not_infer_departments: true do_not_infer_responsibilities: true do_not_infer_attendance: true - mentioned_people_are_not_participants: true \ No newline at end of file + mentioned_people_are_not_participants: true diff --git a/samples/real_live/project_process_meeting/meeting_context.yaml b/samples/real_live/project_process_meeting/meeting_context.yaml index f38896e..408b1ca 100644 --- a/samples/real_live/project_process_meeting/meeting_context.yaml +++ b/samples/real_live/project_process_meeting/meeting_context.yaml @@ -64,7 +64,7 @@ mentioned_people: - "Giovana" role: "Leiterin Business Development" department_id: "bd" - attendance_status: "not_present" + attendance_status: "mentioned_only" notes: null organization: diff --git a/samples/templates/meeting_context.template.yaml b/samples/templates/meeting_context.template.yaml index ac0c9a2..6f2fabd 100644 --- a/samples/templates/meeting_context.template.yaml +++ b/samples/templates/meeting_context.template.yaml @@ -45,7 +45,7 @@ mentioned_people: aliases: [] role: null department: null - attendance_status: "not_present" + attendance_status: "mentioned_only" notes: null organization: diff --git a/scripts/run_mvp_meeting.py b/scripts/run_mvp_meeting.py index 2e54188..10a97e9 100644 --- a/scripts/run_mvp_meeting.py +++ b/scripts/run_mvp_meeting.py @@ -33,6 +33,16 @@ def parse_args(argv: list[str] | None = None) -> argparse.Namespace: parser.add_argument("audio_file", type=Path) parser.add_argument("--whisper-model", type=Path, required=True) parser.add_argument("--whisper-executable", default="whisper-cli") + parser.add_argument("--ffmpeg-executable", default="ffmpeg") + parser.add_argument( + "--audio-normalization", + action=argparse.BooleanOptionalAction, + default=True, + help=( + "Enable FFmpeg loudness normalization during canonical audio preparation " + "(default: enabled)." + ), + ) parser.add_argument("--context", type=Path) parser.add_argument("--output-root", type=Path, default=DEFAULT_OUTPUT_ROOT) parser.add_argument("--language", default="de") @@ -73,6 +83,8 @@ def config_from_args(args: argparse.Namespace) -> MvpMeetingConfig: audio_file=args.audio_file, whisper_model=args.whisper_model, whisper_executable=args.whisper_executable, + ffmpeg_executable=args.ffmpeg_executable, + audio_normalization=args.audio_normalization, context_file=args.context, output_root=args.output_root, language=args.language, diff --git a/src/meeting_lab/audio/__init__.py b/src/meeting_lab/audio/__init__.py new file mode 100644 index 0000000..0b383c9 --- /dev/null +++ b/src/meeting_lab/audio/__init__.py @@ -0,0 +1,5 @@ +"""Canonical audio preparation boundary.""" + +from .preparation import AudioPreparationError, PreparedAudio, prepare_audio + +__all__ = ["AudioPreparationError", "PreparedAudio", "prepare_audio"] diff --git a/src/meeting_lab/audio/preparation.py b/src/meeting_lab/audio/preparation.py new file mode 100644 index 0000000..301098d --- /dev/null +++ b/src/meeting_lab/audio/preparation.py @@ -0,0 +1,190 @@ +"""Prepare supported recordings for deterministic downstream processing.""" + +from __future__ import annotations + +import os +import shutil +import subprocess +import wave +from collections.abc import Callable, Sequence +from dataclasses import dataclass +from pathlib import Path + +SUPPORTED_EXTENSIONS = {".wav", ".flac", ".m4a"} +CANONICAL_SAMPLE_RATE = 16_000 +CANONICAL_CHANNELS = 1 +CANONICAL_SAMPLE_WIDTH_BYTES = 2 +CANONICAL_CODEC = "pcm_s16le" +DEFAULT_NORMALIZATION_FILTER = "loudnorm=I=-16:LRA=11:TP=-1.5" +DEFAULT_NORMALIZATION_METHOD = "ffmpeg_loudnorm" + + +class AudioPreparationError(RuntimeError): + """Raised when source audio cannot be prepared as canonical WAV.""" + + +@dataclass(frozen=True) +class PreparedAudio: + source_path: Path + source_format: str + prepared_path: Path + method: str + ffmpeg_executable: str + normalization_enabled: bool = True + normalization_method: str | None = DEFAULT_NORMALIZATION_METHOD + normalization_filter: str | None = DEFAULT_NORMALIZATION_FILTER + + def metadata(self) -> dict[str, object]: + return { + "original_source_path": str(self.source_path.resolve()), + "original_source_name": self.source_path.name, + "original_format": self.source_format, + "prepared_audio_path": str(self.prepared_path.resolve()), + "preparation_method": self.method, + "ffmpeg_executable": self.ffmpeg_executable, + "normalization_enabled": self.normalization_enabled, + "normalization_method": self.normalization_method, + "normalization_filter": self.normalization_filter, + "canonical_output": { + "container": "wav", + "codec": CANONICAL_CODEC, + "channels": CANONICAL_CHANNELS, + "sample_rate_hz": CANONICAL_SAMPLE_RATE, + "bits_per_sample": CANONICAL_SAMPLE_WIDTH_BYTES * 8, + }, + } + + +Runner = Callable[..., subprocess.CompletedProcess[str]] + + +def prepare_audio( + source_path: Path, + prepared_path: Path, + *, + ffmpeg_executable: str = "ffmpeg", + normalization_enabled: bool = True, + runner: Runner = subprocess.run, +) -> PreparedAudio: + """Create and validate a canonical mono 16 kHz signed PCM16 WAV artifact.""" + source_path = Path(source_path) + prepared_path = Path(prepared_path) + source_format = source_path.suffix.lower() + if not source_path.is_file(): + raise AudioPreparationError(f"Source audio does not exist: {source_path}") + if source_format not in SUPPORTED_EXTENSIONS: + supported = ", ".join(sorted(SUPPORTED_EXTENSIONS)) + raise AudioPreparationError( + f"Unsupported audio format {source_format or ''!r}; supported: {supported}." + ) + + resolved_executable = _resolve_executable(ffmpeg_executable) + prepared_path.parent.mkdir(parents=True, exist_ok=True) + temporary_path = prepared_path.with_name(f".{prepared_path.name}.tmp.wav") + command_parts = [ + resolved_executable, + "-nostdin", + "-hide_banner", + "-loglevel", + "error", + "-y", + "-i", + str(source_path), + "-map_metadata", + "-1", + "-vn", + ] + if normalization_enabled: + command_parts.extend(("-af", DEFAULT_NORMALIZATION_FILTER)) + command_parts.extend( + ( + "-ac", + str(CANONICAL_CHANNELS), + "-ar", + str(CANONICAL_SAMPLE_RATE), + "-c:a", + CANONICAL_CODEC, + "-fflags", + "+bitexact", + str(temporary_path), + ) + ) + command: Sequence[str] = tuple(command_parts) + try: + completed = runner(command, capture_output=True, text=True, check=False) + except OSError as exc: + raise AudioPreparationError(f"Could not run FFmpeg: {exc}") from exc + if completed.returncode != 0: + detail = ( + completed.stderr or completed.stdout or "no diagnostic output" + ).strip() + raise AudioPreparationError( + f"FFmpeg failed to prepare {source_path.name} (exit {completed.returncode}): " + f"{detail}" + ) + try: + _validate_canonical_wav(temporary_path) + os.replace(temporary_path, prepared_path) + except Exception: + temporary_path.unlink(missing_ok=True) + raise + + return PreparedAudio( + source_path=source_path, + source_format=source_format.removeprefix("."), + prepared_path=prepared_path, + method="ffmpeg", + ffmpeg_executable=resolved_executable, + normalization_enabled=normalization_enabled, + normalization_method=( + DEFAULT_NORMALIZATION_METHOD if normalization_enabled else None + ), + normalization_filter=( + DEFAULT_NORMALIZATION_FILTER if normalization_enabled else None + ), + ) + + +def _resolve_executable(executable: str) -> str: + value = executable.strip() + if not value: + raise AudioPreparationError("FFmpeg executable must not be empty.") + if Path(value).parent != Path("."): + path = Path(value) + if path.is_file() and os.access(path, os.X_OK): + return str(path) + raise AudioPreparationError(f"FFmpeg executable is not available: {value}") + resolved = shutil.which(value) + if resolved is None: + raise AudioPreparationError( + f"FFmpeg executable {value!r} was not found on PATH. Install FFmpeg or " + "configure its executable path." + ) + return resolved + + +def _validate_canonical_wav(path: Path) -> None: + try: + with wave.open(str(path), "rb") as recording: + properties = ( + recording.getnchannels(), + recording.getframerate(), + recording.getsampwidth(), + recording.getcomptype(), + ) + except (OSError, EOFError, wave.Error) as exc: + raise AudioPreparationError( + f"FFmpeg did not produce a readable WAV file: {path}: {exc}" + ) from exc + expected = ( + CANONICAL_CHANNELS, + CANONICAL_SAMPLE_RATE, + CANONICAL_SAMPLE_WIDTH_BYTES, + "NONE", + ) + if properties != expected: + raise AudioPreparationError( + "Prepared audio is not canonical mono 16 kHz PCM16 WAV: " + f"channels={properties[0]}, sample_rate={properties[1]}, " + f"sample_width={properties[2]}, compression={properties[3]}." + ) diff --git a/src/meeting_lab/models/meeting_context.py b/src/meeting_lab/models/meeting_context.py index 7acd8ce..a33b474 100644 --- a/src/meeting_lab/models/meeting_context.py +++ b/src/meeting_lab/models/meeting_context.py @@ -11,7 +11,7 @@ from typing import Any SUPPORTED_SCHEMA_VERSIONS = {"1"} -VALID_ATTENDANCE_STATUSES = {"present", "not_present", "absent"} +VALID_ATTENDANCE_STATUSES = {"present", "mentioned_only"} class MeetingContextValidationError(ValueError): @@ -59,15 +59,16 @@ def load_meeting_context(path: Path) -> MeetingContext: if not isinstance(loaded, dict): raise MeetingContextValidationError("Meeting Context must be a YAML object.") - validate_meeting_context(loaded) - return MeetingContext(data=loaded, source_file=path) + normalized = _with_attendance_defaults(loaded) + validate_meeting_context(normalized) + return MeetingContext(data=normalized, source_file=path) def create_meeting_context( data: dict[str, Any], *, source_file: Path = Path("") ) -> MeetingContext: """Validate structured data and return an immutable context boundary.""" - validated = copy.deepcopy(data) + validated = _with_attendance_defaults(data) validate_meeting_context(validated) return MeetingContext(data=validated, source_file=source_file) @@ -143,8 +144,8 @@ def validate_meeting_context(data: dict[str, Any]) -> None: for index, participant in enumerate(participants): item_path = f"participants[{index}]" - _validate_attendance(participant, item_path) - if participant.get("attendance_status") != "present": + status = _validate_attendance(participant, item_path, default="present") + if status != "present": raise MeetingContextValidationError( f"{item_path}.attendance_status must be 'present'." ) @@ -152,10 +153,10 @@ def validate_meeting_context(data: dict[str, Any]) -> None: for index, person in enumerate(mentioned_people): item_path = f"mentioned_people[{index}]" - _validate_attendance(person, item_path) - if person.get("attendance_status") == "present": + status = _validate_attendance(person, item_path, default="mentioned_only") + if status != "mentioned_only": raise MeetingContextValidationError( - f"{item_path}.attendance_status must not be 'present'." + f"{item_path}.attendance_status must be 'mentioned_only'." ) _validate_department_reference(person, item_path, department_ids) @@ -334,12 +335,28 @@ def _collect_unique_ids(items: list[Any], key: str, path: str) -> set[str]: return ids -def _validate_attendance(item: dict[str, Any], path: str) -> None: - status = item.get("attendance_status") +def _validate_attendance(item: dict[str, Any], path: str, *, default: str) -> str: + status = item.get("attendance_status", default) if status not in VALID_ATTENDANCE_STATUSES: raise MeetingContextValidationError( f"{path}.attendance_status has invalid value: {status!r}." ) + return status + + +def _with_attendance_defaults(data: dict[str, Any]) -> dict[str, Any]: + normalized = copy.deepcopy(data) + participants = normalized.get("participants") + if isinstance(participants, list): + for participant in participants: + if isinstance(participant, dict): + participant.setdefault("attendance_status", "present") + mentioned_people = normalized.get("mentioned_people") + if isinstance(mentioned_people, list): + for person in mentioned_people: + if isinstance(person, dict): + person.setdefault("attendance_status", "mentioned_only") + return normalized def _validate_department_reference( diff --git a/src/meeting_lab/orchestration/mvp.py b/src/meeting_lab/orchestration/mvp.py index c4d0c5f..55e3653 100644 --- a/src/meeting_lab/orchestration/mvp.py +++ b/src/meeting_lab/orchestration/mvp.py @@ -7,11 +7,13 @@ import re import shutil import sys import time +from collections.abc import Callable, Mapping, Sequence from dataclasses import dataclass from datetime import datetime from pathlib import Path -from typing import Any, Callable, Mapping, Sequence +from typing import Any +from src.meeting_lab.audio import prepare_audio from src.meeting_lab.diarization import ( DEFAULT_MODEL as DEFAULT_DIARIZATION_MODEL, diarize_audio, @@ -44,6 +46,8 @@ class MvpMeetingConfig: audio_file: Path whisper_model: Path whisper_executable: str = "whisper-cli" + ffmpeg_executable: str = "ffmpeg" + audio_normalization: bool = True context_file: Path | None = None output_root: Path = DEFAULT_OUTPUT_ROOT language: str = "de" @@ -187,6 +191,7 @@ def run_mvp_meeting( stage_runtimes: dict[str, float | None] = { "validation": round(validation_runtime, 3), "setup": None, + "audio_preparation": None, "whisper": None, "transcript_validation": None, "protocol": None, @@ -198,6 +203,7 @@ def run_mvp_meeting( "run_id": run_dir.name, "timestamp": timestamp, "input_audio": str(config.audio_file.resolve()), + "audio_preparation": None, "transcript_output": str(transcript_path.resolve()), "protocol_output": str(protocol_path.resolve()), "whisper_model": str(config.whisper_model.resolve()), @@ -238,6 +244,33 @@ def run_mvp_meeting( }, ) + preparation_started = time.perf_counter() + current_stage = "audio_preparation" + stage_started = preparation_started + prepared_audio = prepare_audio( + config.audio_file, + audio_dir / "prepared.wav", + ffmpeg_executable=config.ffmpeg_executable, + normalization_enabled=config.audio_normalization, + ) + stage_runtimes["audio_preparation"] = round( + time.perf_counter() - preparation_started, 3 + ) + current_stage = "preparing" + preparation_metadata = prepared_audio.metadata() + metadata["audio_preparation"] = preparation_metadata + _write_json(audio_dir / "preparation_metadata.json", preparation_metadata) + _write_json( + audio_dir / "input_manifest.json", + { + "source_file": str(config.audio_file.resolve()), + "filename": config.audio_file.name, + "size_bytes": config.audio_file.stat().st_size, + "format": config.audio_file.suffix.lower().removeprefix("."), + "prepared_audio": preparation_metadata, + }, + ) + preserved_context: Path | None = None if effective_context is not None: preserved_context = context_dir / "meeting_context.yaml" @@ -252,7 +285,7 @@ def run_mvp_meeting( stage_started = time.perf_counter() _emit(progress_sink, "transcription", "started", overall_started) transcription = transcribe_audio( - config.audio_file, + prepared_audio.prepared_path, config.whisper_model, transcript_dir, config.language, @@ -275,7 +308,7 @@ def run_mvp_meeting( _emit(progress_sink, "diarization", "started", overall_started) diarization_dir = run_dir / "diarization" diarization = diarize_audio( - config.audio_file, + prepared_audio.prepared_path, diarization_dir, config.diarization, runtime=config.diarization_runtime, @@ -330,6 +363,7 @@ def run_mvp_meeting( except Exception as exc: metadata_stage = { "preparing": "setup", + "audio_preparation": "audio_preparation", "transcription": "whisper", "diarization": "diarization", "protocol_generation": "protocol", diff --git a/tests/test_audio_preparation.py b/tests/test_audio_preparation.py new file mode 100644 index 0000000..38f35b8 --- /dev/null +++ b/tests/test_audio_preparation.py @@ -0,0 +1,230 @@ +import subprocess +import tempfile +import unittest +import wave +from pathlib import Path +from unittest.mock import patch + +from src.meeting_lab.audio.preparation import ( + DEFAULT_NORMALIZATION_FILTER, + DEFAULT_NORMALIZATION_METHOD, + AudioPreparationError, + prepare_audio, +) + + +def _write_wav( + path: Path, *, channels: int = 1, sample_rate: int = 16_000, sample_width: int = 2 +) -> None: + with wave.open(str(path), "wb") as recording: + recording.setnchannels(channels) + recording.setsampwidth(sample_width) + recording.setframerate(sample_rate) + recording.writeframes(b"\x00" * channels * sample_width * 32) + + +def _successful_runner(commands: list[list[str]]): + def run(command, **kwargs): + commands.append(list(command)) + _write_wav(Path(command[-1])) + return subprocess.CompletedProcess(command, 0, "", "") + + return run + + +class AudioPreparationTests(unittest.TestCase): + def test_supported_inputs_are_prepared_with_normalization_on_and_off(self) -> None: + for suffix in (".wav", ".flac", ".m4a"): + for normalization_enabled in (True, False): + with ( + self.subTest( + suffix=suffix, normalization_enabled=normalization_enabled + ), + tempfile.TemporaryDirectory() as directory, + ): + root = Path(directory) + source = root / f"meeting{suffix}" + if suffix == ".wav": + _write_wav(source) + else: + source.write_bytes(b"original encoded audio") + original = source.read_bytes() + destination = root / "run" / "audio" / "prepared.wav" + commands: list[list[str]] = [] + + with patch( + "src.meeting_lab.audio.preparation.shutil.which", + return_value="/usr/bin/ffmpeg", + ): + result = prepare_audio( + source, + destination, + normalization_enabled=normalization_enabled, + runner=_successful_runner(commands), + ) + + self.assertEqual(source.read_bytes(), original) + self.assertEqual(result.prepared_path, destination) + with wave.open(str(destination), "rb") as recording: + self.assertEqual(recording.getnchannels(), 1) + self.assertEqual(recording.getframerate(), 16_000) + self.assertEqual(recording.getsampwidth(), 2) + self.assertEqual(recording.getcomptype(), "NONE") + self.assertEqual(commands[0][commands[0].index("-ac") + 1], "1") + self.assertEqual(commands[0][commands[0].index("-ar") + 1], "16000") + self.assertEqual( + commands[0][commands[0].index("-c:a") + 1], "pcm_s16le" + ) + self.assertEqual("-af" in commands[0], normalization_enabled) + if normalization_enabled: + self.assertEqual( + commands[0][commands[0].index("-af") + 1], + DEFAULT_NORMALIZATION_FILTER, + ) + self.assertEqual( + result.normalization_enabled, normalization_enabled + ) + + def test_normalization_defaults_to_on_and_explicit_on_matches(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + source = root / "meeting.wav" + _write_wav(source) + commands: list[list[str]] = [] + with patch( + "src.meeting_lab.audio.preparation.shutil.which", + return_value="/usr/bin/ffmpeg", + ): + default = prepare_audio( + source, root / "default.wav", runner=_successful_runner(commands) + ) + explicit = prepare_audio( + source, + root / "explicit.wav", + normalization_enabled=True, + runner=_successful_runner(commands), + ) + + self.assertTrue(default.normalization_enabled) + self.assertTrue(explicit.normalization_enabled) + self.assertEqual( + commands[0][commands[0].index("-af") + 1], + commands[1][commands[1].index("-af") + 1], + ) + + def test_noncanonical_wav_is_normalized(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + source = root / "stereo-48k.wav" + _write_wav(source, channels=2, sample_rate=48_000) + destination = root / "prepared.wav" + + with patch( + "src.meeting_lab.audio.preparation.shutil.which", + return_value="/usr/bin/ffmpeg", + ): + prepare_audio(source, destination, runner=_successful_runner([])) + + with wave.open(str(destination), "rb") as recording: + self.assertEqual( + (recording.getnchannels(), recording.getframerate()), (1, 16_000) + ) + + def test_ffmpeg_missing_has_actionable_error(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + source = root / "meeting.flac" + source.write_bytes(b"audio") + + with ( + patch( + "src.meeting_lab.audio.preparation.shutil.which", return_value=None + ), + self.assertRaisesRegex(AudioPreparationError, "not found on PATH"), + ): + prepare_audio(source, root / "prepared.wav") + + def test_ffmpeg_failure_includes_diagnostic_and_preserves_source(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + source = root / "meeting.m4a" + source.write_bytes(b"original") + + def fail(command, **kwargs): + return subprocess.CompletedProcess(command, 1, "", "decoder exploded") + + with ( + patch( + "src.meeting_lab.audio.preparation.shutil.which", + return_value="/usr/bin/ffmpeg", + ), + self.assertRaisesRegex(AudioPreparationError, "decoder exploded"), + ): + prepare_audio(source, root / "prepared.wav", runner=fail) + + self.assertEqual(source.read_bytes(), b"original") + self.assertFalse((root / "prepared.wav").exists()) + + def test_prepared_audio_metadata_is_traceable(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + source = root / "unknown_meeting.flac" + source.write_bytes(b"source") + destination = root / "audio" / "prepared.wav" + with patch( + "src.meeting_lab.audio.preparation.shutil.which", + return_value="/usr/bin/ffmpeg", + ): + result = prepare_audio( + source, destination, runner=_successful_runner([]) + ) + + metadata = result.metadata() + self.assertEqual(metadata["original_source_name"], "unknown_meeting.flac") + self.assertEqual(metadata["original_format"], "flac") + self.assertEqual( + metadata["prepared_audio_path"], str(destination.resolve()) + ) + self.assertEqual(metadata["preparation_method"], "ffmpeg") + self.assertTrue(metadata["normalization_enabled"]) + self.assertEqual( + metadata["normalization_method"], DEFAULT_NORMALIZATION_METHOD + ) + self.assertEqual( + metadata["normalization_filter"], DEFAULT_NORMALIZATION_FILTER + ) + self.assertEqual( + metadata["canonical_output"], + { + "container": "wav", + "codec": "pcm_s16le", + "channels": 1, + "sample_rate_hz": 16_000, + "bits_per_sample": 16, + }, + ) + + def test_disabled_normalization_metadata_has_no_method_or_filter(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + source = root / "meeting.m4a" + source.write_bytes(b"source") + with patch( + "src.meeting_lab.audio.preparation.shutil.which", + return_value="/usr/bin/ffmpeg", + ): + result = prepare_audio( + source, + root / "prepared.wav", + normalization_enabled=False, + runner=_successful_runner([]), + ) + + metadata = result.metadata() + self.assertFalse(metadata["normalization_enabled"]) + self.assertIsNone(metadata["normalization_method"]) + self.assertIsNone(metadata["normalization_filter"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_meeting_context.py b/tests/test_meeting_context.py index 4b453d1..56d59d3 100644 --- a/tests/test_meeting_context.py +++ b/tests/test_meeting_context.py @@ -72,6 +72,38 @@ class MeetingContextTests(unittest.TestCase): with self.assertRaisesRegex(MeetingContextValidationError, "invalid value"): validate_meeting_context(data) + def test_missing_participant_attendance_defaults_to_present(self) -> None: + data = copy.deepcopy(self.context.data) + del data["participants"][0]["attendance_status"] + + context = create_meeting_context(data) + + self.assertEqual( + context.data["participants"][0]["attendance_status"], "present" + ) + + def test_explicit_mentioned_only_is_preserved(self) -> None: + data = copy.deepcopy(self.context.data) + data["mentioned_people"][0]["attendance_status"] = "mentioned_only" + + context = create_meeting_context(data) + + self.assertEqual( + context.data["mentioned_people"][0]["attendance_status"], + "mentioned_only", + ) + self.assertIn( + "Mentioned but absent people:", render_meeting_context_for_prompt(context) + ) + + def test_mentioned_only_person_cannot_be_a_diarized_speaker(self) -> None: + data = copy.deepcopy(self.context.data) + mentioned_id = data["mentioned_people"][0]["person_id"] + data["speaker_mappings"] = {"SPEAKER_00": mentioned_id} + + with self.assertRaisesRegex(MeetingContextValidationError, "unknown participant"): + validate_meeting_context(data) + def test_prompt_representation_is_deterministic(self) -> None: first = render_meeting_context_for_prompt(self.context) second = render_meeting_context_for_prompt(self.context) diff --git a/tests/test_mvp_api.py b/tests/test_mvp_api.py index 642d3c0..6091ac5 100644 --- a/tests/test_mvp_api.py +++ b/tests/test_mvp_api.py @@ -1,11 +1,14 @@ import json +import shutil import subprocess import tempfile import unittest +from dataclasses import replace from pathlib import Path from unittest.mock import patch from scripts import run_mvp_meeting as cli +from src.meeting_lab.audio import PreparedAudio from src.meeting_lab.models.meeting_context import load_meeting_context from src.meeting_lab.orchestration import mvp as mvp_api from src.meeting_lab.orchestration.mvp import MvpMeetingConfig, MvpRunResult @@ -71,6 +74,12 @@ def fake_transcribe(audio, model, output, language, **kwargs): return TranscriptionResult(output, raw, transcript, text, metadata, 0.1) +def fake_prepare(source, destination, **kwargs): + destination.parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile(source, destination) + return PreparedAudio(source, source.suffix.removeprefix("."), destination, "ffmpeg", "ffmpeg") + + def fake_protocol(transcript, context, **kwargs): rendered_context = load_meeting_context(context) assert rendered_context.meeting_id == "programmatic-test" @@ -103,6 +112,7 @@ class MvpApiTests(unittest.TestCase): events = [] with ( patch.object(mvp_api, "transcribe_audio", side_effect=fake_transcribe), + patch.object(mvp_api, "prepare_audio", side_effect=fake_prepare), patch.object( mvp_api, "generate_direct_protocol", side_effect=fake_protocol ), @@ -142,7 +152,7 @@ class MvpApiTests(unittest.TestCase): mvp_api, "transcribe_audio", side_effect=TranscriptionError("stopped"), - ): + ), patch.object(mvp_api, "prepare_audio", side_effect=fake_prepare): result = mvp_api.run_mvp_meeting(config, progress_sink=events.append) self.assertEqual(result.exit_code, 2) @@ -168,8 +178,72 @@ class MvpApiTests(unittest.TestCase): self.assertEqual(delegated.diarization, "off") self.assertEqual(delegated.language, "de") self.assertEqual(delegated.whisper_executable, "whisper-cli") + self.assertEqual(delegated.ffmpeg_executable, "ffmpeg") + self.assertTrue(delegated.audio_normalization) self.assertEqual(api.call_args.kwargs["meeting_context"], context_data()) + def test_cli_explicit_audio_normalization_values_are_propagated(self): + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + config = self.config(root) + enabled = cli.config_from_args( + cli.parse_args( + [ + str(config.audio_file), + "--whisper-model", + str(config.whisper_model), + "--audio-normalization", + ] + ) + ) + disabled = cli.config_from_args( + cli.parse_args( + [ + str(config.audio_file), + "--whisper-model", + str(config.whisper_model), + "--no-audio-normalization", + ] + ) + ) + + self.assertTrue(enabled.audio_normalization) + self.assertFalse(disabled.audio_normalization) + + def test_transcription_receives_prepared_wav_for_encoded_inputs(self): + for suffix in (".flac", ".m4a"): + with self.subTest(suffix=suffix), tempfile.TemporaryDirectory() as directory: + root = Path(directory) + config = self.config(root) + encoded = config.audio_file.with_suffix(suffix) + config.audio_file.rename(encoded) + config = replace(config, audio_file=encoded) + received = [] + + def capture_transcribe(audio, *args, received_paths=received, **kwargs): + received_paths.append(audio) + return fake_transcribe(audio, *args, **kwargs) + + with ( + patch.object(mvp_api, "prepare_audio", side_effect=fake_prepare), + patch.object(mvp_api, "transcribe_audio", side_effect=capture_transcribe), + patch.object(mvp_api, "generate_direct_protocol", side_effect=fake_protocol), + ): + result = mvp_api.run_mvp_meeting( + config, meeting_context=context_data() + ) + + self.assertEqual(result.exit_code, 0) + self.assertEqual(received, [result.run_dir / "audio" / "prepared.wav"]) + manifest = json.loads( + (result.run_dir / "audio" / "input_manifest.json").read_text() + ) + self.assertEqual(manifest["format"], suffix.removeprefix(".")) + self.assertEqual( + manifest["prepared_audio"]["prepared_audio_path"], + str((result.run_dir / "audio" / "prepared.wav").resolve()), + ) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_mvp_orchestrator.py b/tests/test_mvp_orchestrator.py index eda8b22..86c4043 100644 --- a/tests/test_mvp_orchestrator.py +++ b/tests/test_mvp_orchestrator.py @@ -1,4 +1,5 @@ import json +import shutil import tempfile import unittest from datetime import datetime @@ -6,6 +7,11 @@ from pathlib import Path from unittest.mock import Mock, patch from scripts import run_mvp_meeting +from src.meeting_lab.audio import PreparedAudio +from src.meeting_lab.audio.preparation import ( + DEFAULT_NORMALIZATION_FILTER, + DEFAULT_NORMALIZATION_METHOD, +) from src.meeting_lab.orchestration import mvp as mvp_api from src.meeting_lab.protocol.generate_direct_protocol import DirectProtocolResult from src.meeting_lab.diarization.backend import DiarizationResult @@ -106,7 +112,28 @@ def fake_diarize(audio_path, output_dir, device_mode, **kwargs): ) +def fake_prepare(source, destination, **kwargs): + destination.parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile(source, destination) + normalization_enabled = kwargs.get("normalization_enabled", True) + return PreparedAudio( + source, + source.suffix.removeprefix("."), + destination, + "ffmpeg", + kwargs.get("ffmpeg_executable", "ffmpeg"), + normalization_enabled, + DEFAULT_NORMALIZATION_METHOD if normalization_enabled else None, + DEFAULT_NORMALIZATION_FILTER if normalization_enabled else None, + ) + + class MvpOrchestratorTests(unittest.TestCase): + def setUp(self) -> None: + patcher = patch.object(mvp_api, "prepare_audio", side_effect=fake_prepare) + self.prepare_audio = patcher.start() + self.addCleanup(patcher.stop) + def create_inputs(self, root: Path) -> tuple[Path, Path, Path]: audio = root / "team meeting.wav" whisper_model = root / "ggml-model.bin" @@ -151,6 +178,8 @@ class MvpOrchestratorTests(unittest.TestCase): expected = { "run_metadata.json", "audio/input_manifest.json", + "audio/prepared.wav", + "audio/preparation_metadata.json", "transcript/whisper_raw.json", "transcript/transcript.json", "transcript/transcript.txt", @@ -167,14 +196,60 @@ class MvpOrchestratorTests(unittest.TestCase): self.assertEqual(metadata["model"], "chosen:model") self.assertIsNone(metadata["failure"]) self.assertEqual(whisper.call_count, 1) + self.assertEqual( + whisper.call_args.args[0], run_dir / "audio" / "prepared.wav" + ) self.assertEqual(protocol.call_count, 1) + self.assertTrue( + self.prepare_audio.call_args.kwargs["normalization_enabled"] + ) + preparation = json.loads( + (run_dir / "audio" / "preparation_metadata.json").read_text() + ) + self.assertTrue(preparation["normalization_enabled"]) + + def test_cli_can_disable_audio_normalization_without_bypassing_preparation(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = Path(directory) + args = self.args(root, ["--no-audio-normalization"]) + with ( + patch.object(mvp_api, "transcribe_audio", side_effect=fake_transcribe) as whisper, + patch.object( + mvp_api, + "generate_direct_protocol", + return_value=protocol_result(), + ), + ): + code, run_dir, _ = run_mvp_meeting.run(args) + + self.assertEqual(code, 0) + self.prepare_audio.assert_called_once() + self.assertFalse( + self.prepare_audio.call_args.kwargs["normalization_enabled"] + ) + self.assertEqual( + whisper.call_args.args[0], run_dir / "audio" / "prepared.wav" + ) + preparation = json.loads( + (run_dir / "audio" / "preparation_metadata.json").read_text() + ) + self.assertFalse(preparation["normalization_enabled"]) + self.assertIsNone(preparation["normalization_method"]) + self.assertIsNone(preparation["normalization_filter"]) def test_context_model_endpoint_and_whisper_options_are_forwarded(self) -> None: with tempfile.TemporaryDirectory() as directory: root = Path(directory) args = self.args( root, - ["--whisper-executable", "/tools/whisper-cli", "--threads", "4"], + [ + "--whisper-executable", + "/tools/whisper-cli", + "--ffmpeg-executable", + "/tools/ffmpeg", + "--threads", + "4", + ], ) with ( patch.object(mvp_api, "transcribe_audio", side_effect=fake_transcribe) as whisper, @@ -190,6 +265,10 @@ class MvpOrchestratorTests(unittest.TestCase): self.assertEqual(whisper.call_args.args[3], "de") self.assertEqual(whisper.call_args.kwargs["executable"], "/tools/whisper-cli") self.assertEqual(whisper.call_args.kwargs["threads"], "4") + self.assertEqual( + self.prepare_audio.call_args.kwargs["ffmpeg_executable"], + "/tools/ffmpeg", + ) self.assertEqual(protocol.call_args.args[1], run_dir / "context/meeting_context.yaml") self.assertEqual(protocol.call_args.kwargs["model"], "chosen:model") self.assertEqual( @@ -245,6 +324,9 @@ class MvpOrchestratorTests(unittest.TestCase): self.assertEqual(code, 0) self.assertEqual(diarization.call_args.args[2], "gpu") + self.assertEqual( + diarization.call_args.args[0], run_dir / "audio" / "prepared.wav" + ) self.assertEqual(diarization.call_args.kwargs["runtime"], "container") self.assertEqual( diarization.call_args.kwargs["container_args"], ("--device=/dev/kfd",)