Add canonical audio preparation and meeting context support
This commit is contained in:
@@ -0,0 +1,5 @@
|
||||
"""Canonical audio preparation boundary."""
|
||||
|
||||
from .preparation import AudioPreparationError, PreparedAudio, prepare_audio
|
||||
|
||||
__all__ = ["AudioPreparationError", "PreparedAudio", "prepare_audio"]
|
||||
@@ -0,0 +1,190 @@
|
||||
"""Prepare supported recordings for deterministic downstream processing."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import wave
|
||||
from collections.abc import Callable, Sequence
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
SUPPORTED_EXTENSIONS = {".wav", ".flac", ".m4a"}
|
||||
CANONICAL_SAMPLE_RATE = 16_000
|
||||
CANONICAL_CHANNELS = 1
|
||||
CANONICAL_SAMPLE_WIDTH_BYTES = 2
|
||||
CANONICAL_CODEC = "pcm_s16le"
|
||||
DEFAULT_NORMALIZATION_FILTER = "loudnorm=I=-16:LRA=11:TP=-1.5"
|
||||
DEFAULT_NORMALIZATION_METHOD = "ffmpeg_loudnorm"
|
||||
|
||||
|
||||
class AudioPreparationError(RuntimeError):
|
||||
"""Raised when source audio cannot be prepared as canonical WAV."""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PreparedAudio:
|
||||
source_path: Path
|
||||
source_format: str
|
||||
prepared_path: Path
|
||||
method: str
|
||||
ffmpeg_executable: str
|
||||
normalization_enabled: bool = True
|
||||
normalization_method: str | None = DEFAULT_NORMALIZATION_METHOD
|
||||
normalization_filter: str | None = DEFAULT_NORMALIZATION_FILTER
|
||||
|
||||
def metadata(self) -> dict[str, object]:
|
||||
return {
|
||||
"original_source_path": str(self.source_path.resolve()),
|
||||
"original_source_name": self.source_path.name,
|
||||
"original_format": self.source_format,
|
||||
"prepared_audio_path": str(self.prepared_path.resolve()),
|
||||
"preparation_method": self.method,
|
||||
"ffmpeg_executable": self.ffmpeg_executable,
|
||||
"normalization_enabled": self.normalization_enabled,
|
||||
"normalization_method": self.normalization_method,
|
||||
"normalization_filter": self.normalization_filter,
|
||||
"canonical_output": {
|
||||
"container": "wav",
|
||||
"codec": CANONICAL_CODEC,
|
||||
"channels": CANONICAL_CHANNELS,
|
||||
"sample_rate_hz": CANONICAL_SAMPLE_RATE,
|
||||
"bits_per_sample": CANONICAL_SAMPLE_WIDTH_BYTES * 8,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
Runner = Callable[..., subprocess.CompletedProcess[str]]
|
||||
|
||||
|
||||
def prepare_audio(
|
||||
source_path: Path,
|
||||
prepared_path: Path,
|
||||
*,
|
||||
ffmpeg_executable: str = "ffmpeg",
|
||||
normalization_enabled: bool = True,
|
||||
runner: Runner = subprocess.run,
|
||||
) -> PreparedAudio:
|
||||
"""Create and validate a canonical mono 16 kHz signed PCM16 WAV artifact."""
|
||||
source_path = Path(source_path)
|
||||
prepared_path = Path(prepared_path)
|
||||
source_format = source_path.suffix.lower()
|
||||
if not source_path.is_file():
|
||||
raise AudioPreparationError(f"Source audio does not exist: {source_path}")
|
||||
if source_format not in SUPPORTED_EXTENSIONS:
|
||||
supported = ", ".join(sorted(SUPPORTED_EXTENSIONS))
|
||||
raise AudioPreparationError(
|
||||
f"Unsupported audio format {source_format or '<none>'!r}; supported: {supported}."
|
||||
)
|
||||
|
||||
resolved_executable = _resolve_executable(ffmpeg_executable)
|
||||
prepared_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
temporary_path = prepared_path.with_name(f".{prepared_path.name}.tmp.wav")
|
||||
command_parts = [
|
||||
resolved_executable,
|
||||
"-nostdin",
|
||||
"-hide_banner",
|
||||
"-loglevel",
|
||||
"error",
|
||||
"-y",
|
||||
"-i",
|
||||
str(source_path),
|
||||
"-map_metadata",
|
||||
"-1",
|
||||
"-vn",
|
||||
]
|
||||
if normalization_enabled:
|
||||
command_parts.extend(("-af", DEFAULT_NORMALIZATION_FILTER))
|
||||
command_parts.extend(
|
||||
(
|
||||
"-ac",
|
||||
str(CANONICAL_CHANNELS),
|
||||
"-ar",
|
||||
str(CANONICAL_SAMPLE_RATE),
|
||||
"-c:a",
|
||||
CANONICAL_CODEC,
|
||||
"-fflags",
|
||||
"+bitexact",
|
||||
str(temporary_path),
|
||||
)
|
||||
)
|
||||
command: Sequence[str] = tuple(command_parts)
|
||||
try:
|
||||
completed = runner(command, capture_output=True, text=True, check=False)
|
||||
except OSError as exc:
|
||||
raise AudioPreparationError(f"Could not run FFmpeg: {exc}") from exc
|
||||
if completed.returncode != 0:
|
||||
detail = (
|
||||
completed.stderr or completed.stdout or "no diagnostic output"
|
||||
).strip()
|
||||
raise AudioPreparationError(
|
||||
f"FFmpeg failed to prepare {source_path.name} (exit {completed.returncode}): "
|
||||
f"{detail}"
|
||||
)
|
||||
try:
|
||||
_validate_canonical_wav(temporary_path)
|
||||
os.replace(temporary_path, prepared_path)
|
||||
except Exception:
|
||||
temporary_path.unlink(missing_ok=True)
|
||||
raise
|
||||
|
||||
return PreparedAudio(
|
||||
source_path=source_path,
|
||||
source_format=source_format.removeprefix("."),
|
||||
prepared_path=prepared_path,
|
||||
method="ffmpeg",
|
||||
ffmpeg_executable=resolved_executable,
|
||||
normalization_enabled=normalization_enabled,
|
||||
normalization_method=(
|
||||
DEFAULT_NORMALIZATION_METHOD if normalization_enabled else None
|
||||
),
|
||||
normalization_filter=(
|
||||
DEFAULT_NORMALIZATION_FILTER if normalization_enabled else None
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _resolve_executable(executable: str) -> str:
|
||||
value = executable.strip()
|
||||
if not value:
|
||||
raise AudioPreparationError("FFmpeg executable must not be empty.")
|
||||
if Path(value).parent != Path("."):
|
||||
path = Path(value)
|
||||
if path.is_file() and os.access(path, os.X_OK):
|
||||
return str(path)
|
||||
raise AudioPreparationError(f"FFmpeg executable is not available: {value}")
|
||||
resolved = shutil.which(value)
|
||||
if resolved is None:
|
||||
raise AudioPreparationError(
|
||||
f"FFmpeg executable {value!r} was not found on PATH. Install FFmpeg or "
|
||||
"configure its executable path."
|
||||
)
|
||||
return resolved
|
||||
|
||||
|
||||
def _validate_canonical_wav(path: Path) -> None:
|
||||
try:
|
||||
with wave.open(str(path), "rb") as recording:
|
||||
properties = (
|
||||
recording.getnchannels(),
|
||||
recording.getframerate(),
|
||||
recording.getsampwidth(),
|
||||
recording.getcomptype(),
|
||||
)
|
||||
except (OSError, EOFError, wave.Error) as exc:
|
||||
raise AudioPreparationError(
|
||||
f"FFmpeg did not produce a readable WAV file: {path}: {exc}"
|
||||
) from exc
|
||||
expected = (
|
||||
CANONICAL_CHANNELS,
|
||||
CANONICAL_SAMPLE_RATE,
|
||||
CANONICAL_SAMPLE_WIDTH_BYTES,
|
||||
"NONE",
|
||||
)
|
||||
if properties != expected:
|
||||
raise AudioPreparationError(
|
||||
"Prepared audio is not canonical mono 16 kHz PCM16 WAV: "
|
||||
f"channels={properties[0]}, sample_rate={properties[1]}, "
|
||||
f"sample_width={properties[2]}, compression={properties[3]}."
|
||||
)
|
||||
@@ -11,7 +11,7 @@ from typing import Any
|
||||
|
||||
|
||||
SUPPORTED_SCHEMA_VERSIONS = {"1"}
|
||||
VALID_ATTENDANCE_STATUSES = {"present", "not_present", "absent"}
|
||||
VALID_ATTENDANCE_STATUSES = {"present", "mentioned_only"}
|
||||
|
||||
|
||||
class MeetingContextValidationError(ValueError):
|
||||
@@ -59,15 +59,16 @@ def load_meeting_context(path: Path) -> MeetingContext:
|
||||
if not isinstance(loaded, dict):
|
||||
raise MeetingContextValidationError("Meeting Context must be a YAML object.")
|
||||
|
||||
validate_meeting_context(loaded)
|
||||
return MeetingContext(data=loaded, source_file=path)
|
||||
normalized = _with_attendance_defaults(loaded)
|
||||
validate_meeting_context(normalized)
|
||||
return MeetingContext(data=normalized, source_file=path)
|
||||
|
||||
|
||||
def create_meeting_context(
|
||||
data: dict[str, Any], *, source_file: Path = Path("<generated>")
|
||||
) -> MeetingContext:
|
||||
"""Validate structured data and return an immutable context boundary."""
|
||||
validated = copy.deepcopy(data)
|
||||
validated = _with_attendance_defaults(data)
|
||||
validate_meeting_context(validated)
|
||||
return MeetingContext(data=validated, source_file=source_file)
|
||||
|
||||
@@ -143,8 +144,8 @@ def validate_meeting_context(data: dict[str, Any]) -> None:
|
||||
|
||||
for index, participant in enumerate(participants):
|
||||
item_path = f"participants[{index}]"
|
||||
_validate_attendance(participant, item_path)
|
||||
if participant.get("attendance_status") != "present":
|
||||
status = _validate_attendance(participant, item_path, default="present")
|
||||
if status != "present":
|
||||
raise MeetingContextValidationError(
|
||||
f"{item_path}.attendance_status must be 'present'."
|
||||
)
|
||||
@@ -152,10 +153,10 @@ def validate_meeting_context(data: dict[str, Any]) -> None:
|
||||
|
||||
for index, person in enumerate(mentioned_people):
|
||||
item_path = f"mentioned_people[{index}]"
|
||||
_validate_attendance(person, item_path)
|
||||
if person.get("attendance_status") == "present":
|
||||
status = _validate_attendance(person, item_path, default="mentioned_only")
|
||||
if status != "mentioned_only":
|
||||
raise MeetingContextValidationError(
|
||||
f"{item_path}.attendance_status must not be 'present'."
|
||||
f"{item_path}.attendance_status must be 'mentioned_only'."
|
||||
)
|
||||
_validate_department_reference(person, item_path, department_ids)
|
||||
|
||||
@@ -334,12 +335,28 @@ def _collect_unique_ids(items: list[Any], key: str, path: str) -> set[str]:
|
||||
return ids
|
||||
|
||||
|
||||
def _validate_attendance(item: dict[str, Any], path: str) -> None:
|
||||
status = item.get("attendance_status")
|
||||
def _validate_attendance(item: dict[str, Any], path: str, *, default: str) -> str:
|
||||
status = item.get("attendance_status", default)
|
||||
if status not in VALID_ATTENDANCE_STATUSES:
|
||||
raise MeetingContextValidationError(
|
||||
f"{path}.attendance_status has invalid value: {status!r}."
|
||||
)
|
||||
return status
|
||||
|
||||
|
||||
def _with_attendance_defaults(data: dict[str, Any]) -> dict[str, Any]:
|
||||
normalized = copy.deepcopy(data)
|
||||
participants = normalized.get("participants")
|
||||
if isinstance(participants, list):
|
||||
for participant in participants:
|
||||
if isinstance(participant, dict):
|
||||
participant.setdefault("attendance_status", "present")
|
||||
mentioned_people = normalized.get("mentioned_people")
|
||||
if isinstance(mentioned_people, list):
|
||||
for person in mentioned_people:
|
||||
if isinstance(person, dict):
|
||||
person.setdefault("attendance_status", "mentioned_only")
|
||||
return normalized
|
||||
|
||||
|
||||
def _validate_department_reference(
|
||||
|
||||
@@ -7,11 +7,13 @@ import re
|
||||
import shutil
|
||||
import sys
|
||||
import time
|
||||
from collections.abc import Callable, Mapping, Sequence
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable, Mapping, Sequence
|
||||
from typing import Any
|
||||
|
||||
from src.meeting_lab.audio import prepare_audio
|
||||
from src.meeting_lab.diarization import (
|
||||
DEFAULT_MODEL as DEFAULT_DIARIZATION_MODEL,
|
||||
diarize_audio,
|
||||
@@ -44,6 +46,8 @@ class MvpMeetingConfig:
|
||||
audio_file: Path
|
||||
whisper_model: Path
|
||||
whisper_executable: str = "whisper-cli"
|
||||
ffmpeg_executable: str = "ffmpeg"
|
||||
audio_normalization: bool = True
|
||||
context_file: Path | None = None
|
||||
output_root: Path = DEFAULT_OUTPUT_ROOT
|
||||
language: str = "de"
|
||||
@@ -187,6 +191,7 @@ def run_mvp_meeting(
|
||||
stage_runtimes: dict[str, float | None] = {
|
||||
"validation": round(validation_runtime, 3),
|
||||
"setup": None,
|
||||
"audio_preparation": None,
|
||||
"whisper": None,
|
||||
"transcript_validation": None,
|
||||
"protocol": None,
|
||||
@@ -198,6 +203,7 @@ def run_mvp_meeting(
|
||||
"run_id": run_dir.name,
|
||||
"timestamp": timestamp,
|
||||
"input_audio": str(config.audio_file.resolve()),
|
||||
"audio_preparation": None,
|
||||
"transcript_output": str(transcript_path.resolve()),
|
||||
"protocol_output": str(protocol_path.resolve()),
|
||||
"whisper_model": str(config.whisper_model.resolve()),
|
||||
@@ -238,6 +244,33 @@ def run_mvp_meeting(
|
||||
},
|
||||
)
|
||||
|
||||
preparation_started = time.perf_counter()
|
||||
current_stage = "audio_preparation"
|
||||
stage_started = preparation_started
|
||||
prepared_audio = prepare_audio(
|
||||
config.audio_file,
|
||||
audio_dir / "prepared.wav",
|
||||
ffmpeg_executable=config.ffmpeg_executable,
|
||||
normalization_enabled=config.audio_normalization,
|
||||
)
|
||||
stage_runtimes["audio_preparation"] = round(
|
||||
time.perf_counter() - preparation_started, 3
|
||||
)
|
||||
current_stage = "preparing"
|
||||
preparation_metadata = prepared_audio.metadata()
|
||||
metadata["audio_preparation"] = preparation_metadata
|
||||
_write_json(audio_dir / "preparation_metadata.json", preparation_metadata)
|
||||
_write_json(
|
||||
audio_dir / "input_manifest.json",
|
||||
{
|
||||
"source_file": str(config.audio_file.resolve()),
|
||||
"filename": config.audio_file.name,
|
||||
"size_bytes": config.audio_file.stat().st_size,
|
||||
"format": config.audio_file.suffix.lower().removeprefix("."),
|
||||
"prepared_audio": preparation_metadata,
|
||||
},
|
||||
)
|
||||
|
||||
preserved_context: Path | None = None
|
||||
if effective_context is not None:
|
||||
preserved_context = context_dir / "meeting_context.yaml"
|
||||
@@ -252,7 +285,7 @@ def run_mvp_meeting(
|
||||
stage_started = time.perf_counter()
|
||||
_emit(progress_sink, "transcription", "started", overall_started)
|
||||
transcription = transcribe_audio(
|
||||
config.audio_file,
|
||||
prepared_audio.prepared_path,
|
||||
config.whisper_model,
|
||||
transcript_dir,
|
||||
config.language,
|
||||
@@ -275,7 +308,7 @@ def run_mvp_meeting(
|
||||
_emit(progress_sink, "diarization", "started", overall_started)
|
||||
diarization_dir = run_dir / "diarization"
|
||||
diarization = diarize_audio(
|
||||
config.audio_file,
|
||||
prepared_audio.prepared_path,
|
||||
diarization_dir,
|
||||
config.diarization,
|
||||
runtime=config.diarization_runtime,
|
||||
@@ -330,6 +363,7 @@ def run_mvp_meeting(
|
||||
except Exception as exc:
|
||||
metadata_stage = {
|
||||
"preparing": "setup",
|
||||
"audio_preparation": "audio_preparation",
|
||||
"transcription": "whisper",
|
||||
"diarization": "diarization",
|
||||
"protocol_generation": "protocol",
|
||||
|
||||
Reference in New Issue
Block a user