Add canonical audio preparation and meeting context support

This commit is contained in:
2026-08-24 10:22:16 +02:00
parent 8dab928763
commit d94436af43
14 changed files with 713 additions and 23 deletions
+5
View File
@@ -0,0 +1,5 @@
"""Canonical audio preparation boundary."""
from .preparation import AudioPreparationError, PreparedAudio, prepare_audio
__all__ = ["AudioPreparationError", "PreparedAudio", "prepare_audio"]
+190
View File
@@ -0,0 +1,190 @@
"""Prepare supported recordings for deterministic downstream processing."""
from __future__ import annotations
import os
import shutil
import subprocess
import wave
from collections.abc import Callable, Sequence
from dataclasses import dataclass
from pathlib import Path
SUPPORTED_EXTENSIONS = {".wav", ".flac", ".m4a"}
CANONICAL_SAMPLE_RATE = 16_000
CANONICAL_CHANNELS = 1
CANONICAL_SAMPLE_WIDTH_BYTES = 2
CANONICAL_CODEC = "pcm_s16le"
DEFAULT_NORMALIZATION_FILTER = "loudnorm=I=-16:LRA=11:TP=-1.5"
DEFAULT_NORMALIZATION_METHOD = "ffmpeg_loudnorm"
class AudioPreparationError(RuntimeError):
"""Raised when source audio cannot be prepared as canonical WAV."""
@dataclass(frozen=True)
class PreparedAudio:
source_path: Path
source_format: str
prepared_path: Path
method: str
ffmpeg_executable: str
normalization_enabled: bool = True
normalization_method: str | None = DEFAULT_NORMALIZATION_METHOD
normalization_filter: str | None = DEFAULT_NORMALIZATION_FILTER
def metadata(self) -> dict[str, object]:
return {
"original_source_path": str(self.source_path.resolve()),
"original_source_name": self.source_path.name,
"original_format": self.source_format,
"prepared_audio_path": str(self.prepared_path.resolve()),
"preparation_method": self.method,
"ffmpeg_executable": self.ffmpeg_executable,
"normalization_enabled": self.normalization_enabled,
"normalization_method": self.normalization_method,
"normalization_filter": self.normalization_filter,
"canonical_output": {
"container": "wav",
"codec": CANONICAL_CODEC,
"channels": CANONICAL_CHANNELS,
"sample_rate_hz": CANONICAL_SAMPLE_RATE,
"bits_per_sample": CANONICAL_SAMPLE_WIDTH_BYTES * 8,
},
}
Runner = Callable[..., subprocess.CompletedProcess[str]]
def prepare_audio(
source_path: Path,
prepared_path: Path,
*,
ffmpeg_executable: str = "ffmpeg",
normalization_enabled: bool = True,
runner: Runner = subprocess.run,
) -> PreparedAudio:
"""Create and validate a canonical mono 16 kHz signed PCM16 WAV artifact."""
source_path = Path(source_path)
prepared_path = Path(prepared_path)
source_format = source_path.suffix.lower()
if not source_path.is_file():
raise AudioPreparationError(f"Source audio does not exist: {source_path}")
if source_format not in SUPPORTED_EXTENSIONS:
supported = ", ".join(sorted(SUPPORTED_EXTENSIONS))
raise AudioPreparationError(
f"Unsupported audio format {source_format or '<none>'!r}; supported: {supported}."
)
resolved_executable = _resolve_executable(ffmpeg_executable)
prepared_path.parent.mkdir(parents=True, exist_ok=True)
temporary_path = prepared_path.with_name(f".{prepared_path.name}.tmp.wav")
command_parts = [
resolved_executable,
"-nostdin",
"-hide_banner",
"-loglevel",
"error",
"-y",
"-i",
str(source_path),
"-map_metadata",
"-1",
"-vn",
]
if normalization_enabled:
command_parts.extend(("-af", DEFAULT_NORMALIZATION_FILTER))
command_parts.extend(
(
"-ac",
str(CANONICAL_CHANNELS),
"-ar",
str(CANONICAL_SAMPLE_RATE),
"-c:a",
CANONICAL_CODEC,
"-fflags",
"+bitexact",
str(temporary_path),
)
)
command: Sequence[str] = tuple(command_parts)
try:
completed = runner(command, capture_output=True, text=True, check=False)
except OSError as exc:
raise AudioPreparationError(f"Could not run FFmpeg: {exc}") from exc
if completed.returncode != 0:
detail = (
completed.stderr or completed.stdout or "no diagnostic output"
).strip()
raise AudioPreparationError(
f"FFmpeg failed to prepare {source_path.name} (exit {completed.returncode}): "
f"{detail}"
)
try:
_validate_canonical_wav(temporary_path)
os.replace(temporary_path, prepared_path)
except Exception:
temporary_path.unlink(missing_ok=True)
raise
return PreparedAudio(
source_path=source_path,
source_format=source_format.removeprefix("."),
prepared_path=prepared_path,
method="ffmpeg",
ffmpeg_executable=resolved_executable,
normalization_enabled=normalization_enabled,
normalization_method=(
DEFAULT_NORMALIZATION_METHOD if normalization_enabled else None
),
normalization_filter=(
DEFAULT_NORMALIZATION_FILTER if normalization_enabled else None
),
)
def _resolve_executable(executable: str) -> str:
value = executable.strip()
if not value:
raise AudioPreparationError("FFmpeg executable must not be empty.")
if Path(value).parent != Path("."):
path = Path(value)
if path.is_file() and os.access(path, os.X_OK):
return str(path)
raise AudioPreparationError(f"FFmpeg executable is not available: {value}")
resolved = shutil.which(value)
if resolved is None:
raise AudioPreparationError(
f"FFmpeg executable {value!r} was not found on PATH. Install FFmpeg or "
"configure its executable path."
)
return resolved
def _validate_canonical_wav(path: Path) -> None:
try:
with wave.open(str(path), "rb") as recording:
properties = (
recording.getnchannels(),
recording.getframerate(),
recording.getsampwidth(),
recording.getcomptype(),
)
except (OSError, EOFError, wave.Error) as exc:
raise AudioPreparationError(
f"FFmpeg did not produce a readable WAV file: {path}: {exc}"
) from exc
expected = (
CANONICAL_CHANNELS,
CANONICAL_SAMPLE_RATE,
CANONICAL_SAMPLE_WIDTH_BYTES,
"NONE",
)
if properties != expected:
raise AudioPreparationError(
"Prepared audio is not canonical mono 16 kHz PCM16 WAV: "
f"channels={properties[0]}, sample_rate={properties[1]}, "
f"sample_width={properties[2]}, compression={properties[3]}."
)
+28 -11
View File
@@ -11,7 +11,7 @@ from typing import Any
SUPPORTED_SCHEMA_VERSIONS = {"1"}
VALID_ATTENDANCE_STATUSES = {"present", "not_present", "absent"}
VALID_ATTENDANCE_STATUSES = {"present", "mentioned_only"}
class MeetingContextValidationError(ValueError):
@@ -59,15 +59,16 @@ def load_meeting_context(path: Path) -> MeetingContext:
if not isinstance(loaded, dict):
raise MeetingContextValidationError("Meeting Context must be a YAML object.")
validate_meeting_context(loaded)
return MeetingContext(data=loaded, source_file=path)
normalized = _with_attendance_defaults(loaded)
validate_meeting_context(normalized)
return MeetingContext(data=normalized, source_file=path)
def create_meeting_context(
data: dict[str, Any], *, source_file: Path = Path("<generated>")
) -> MeetingContext:
"""Validate structured data and return an immutable context boundary."""
validated = copy.deepcopy(data)
validated = _with_attendance_defaults(data)
validate_meeting_context(validated)
return MeetingContext(data=validated, source_file=source_file)
@@ -143,8 +144,8 @@ def validate_meeting_context(data: dict[str, Any]) -> None:
for index, participant in enumerate(participants):
item_path = f"participants[{index}]"
_validate_attendance(participant, item_path)
if participant.get("attendance_status") != "present":
status = _validate_attendance(participant, item_path, default="present")
if status != "present":
raise MeetingContextValidationError(
f"{item_path}.attendance_status must be 'present'."
)
@@ -152,10 +153,10 @@ def validate_meeting_context(data: dict[str, Any]) -> None:
for index, person in enumerate(mentioned_people):
item_path = f"mentioned_people[{index}]"
_validate_attendance(person, item_path)
if person.get("attendance_status") == "present":
status = _validate_attendance(person, item_path, default="mentioned_only")
if status != "mentioned_only":
raise MeetingContextValidationError(
f"{item_path}.attendance_status must not be 'present'."
f"{item_path}.attendance_status must be 'mentioned_only'."
)
_validate_department_reference(person, item_path, department_ids)
@@ -334,12 +335,28 @@ def _collect_unique_ids(items: list[Any], key: str, path: str) -> set[str]:
return ids
def _validate_attendance(item: dict[str, Any], path: str) -> None:
status = item.get("attendance_status")
def _validate_attendance(item: dict[str, Any], path: str, *, default: str) -> str:
status = item.get("attendance_status", default)
if status not in VALID_ATTENDANCE_STATUSES:
raise MeetingContextValidationError(
f"{path}.attendance_status has invalid value: {status!r}."
)
return status
def _with_attendance_defaults(data: dict[str, Any]) -> dict[str, Any]:
normalized = copy.deepcopy(data)
participants = normalized.get("participants")
if isinstance(participants, list):
for participant in participants:
if isinstance(participant, dict):
participant.setdefault("attendance_status", "present")
mentioned_people = normalized.get("mentioned_people")
if isinstance(mentioned_people, list):
for person in mentioned_people:
if isinstance(person, dict):
person.setdefault("attendance_status", "mentioned_only")
return normalized
def _validate_department_reference(
+37 -3
View File
@@ -7,11 +7,13 @@ import re
import shutil
import sys
import time
from collections.abc import Callable, Mapping, Sequence
from dataclasses import dataclass
from datetime import datetime
from pathlib import Path
from typing import Any, Callable, Mapping, Sequence
from typing import Any
from src.meeting_lab.audio import prepare_audio
from src.meeting_lab.diarization import (
DEFAULT_MODEL as DEFAULT_DIARIZATION_MODEL,
diarize_audio,
@@ -44,6 +46,8 @@ class MvpMeetingConfig:
audio_file: Path
whisper_model: Path
whisper_executable: str = "whisper-cli"
ffmpeg_executable: str = "ffmpeg"
audio_normalization: bool = True
context_file: Path | None = None
output_root: Path = DEFAULT_OUTPUT_ROOT
language: str = "de"
@@ -187,6 +191,7 @@ def run_mvp_meeting(
stage_runtimes: dict[str, float | None] = {
"validation": round(validation_runtime, 3),
"setup": None,
"audio_preparation": None,
"whisper": None,
"transcript_validation": None,
"protocol": None,
@@ -198,6 +203,7 @@ def run_mvp_meeting(
"run_id": run_dir.name,
"timestamp": timestamp,
"input_audio": str(config.audio_file.resolve()),
"audio_preparation": None,
"transcript_output": str(transcript_path.resolve()),
"protocol_output": str(protocol_path.resolve()),
"whisper_model": str(config.whisper_model.resolve()),
@@ -238,6 +244,33 @@ def run_mvp_meeting(
},
)
preparation_started = time.perf_counter()
current_stage = "audio_preparation"
stage_started = preparation_started
prepared_audio = prepare_audio(
config.audio_file,
audio_dir / "prepared.wav",
ffmpeg_executable=config.ffmpeg_executable,
normalization_enabled=config.audio_normalization,
)
stage_runtimes["audio_preparation"] = round(
time.perf_counter() - preparation_started, 3
)
current_stage = "preparing"
preparation_metadata = prepared_audio.metadata()
metadata["audio_preparation"] = preparation_metadata
_write_json(audio_dir / "preparation_metadata.json", preparation_metadata)
_write_json(
audio_dir / "input_manifest.json",
{
"source_file": str(config.audio_file.resolve()),
"filename": config.audio_file.name,
"size_bytes": config.audio_file.stat().st_size,
"format": config.audio_file.suffix.lower().removeprefix("."),
"prepared_audio": preparation_metadata,
},
)
preserved_context: Path | None = None
if effective_context is not None:
preserved_context = context_dir / "meeting_context.yaml"
@@ -252,7 +285,7 @@ def run_mvp_meeting(
stage_started = time.perf_counter()
_emit(progress_sink, "transcription", "started", overall_started)
transcription = transcribe_audio(
config.audio_file,
prepared_audio.prepared_path,
config.whisper_model,
transcript_dir,
config.language,
@@ -275,7 +308,7 @@ def run_mvp_meeting(
_emit(progress_sink, "diarization", "started", overall_started)
diarization_dir = run_dir / "diarization"
diarization = diarize_audio(
config.audio_file,
prepared_audio.prepared_path,
diarization_dir,
config.diarization,
runtime=config.diarization_runtime,
@@ -330,6 +363,7 @@ def run_mvp_meeting(
except Exception as exc:
metadata_stage = {
"preparing": "setup",
"audio_preparation": "audio_preparation",
"transcription": "whisper",
"diarization": "diarization",
"protocol_generation": "protocol",