Add canonical audio preparation and meeting context support
This commit is contained in:
@@ -0,0 +1,230 @@
|
||||
import subprocess
|
||||
import tempfile
|
||||
import unittest
|
||||
import wave
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
from src.meeting_lab.audio.preparation import (
|
||||
DEFAULT_NORMALIZATION_FILTER,
|
||||
DEFAULT_NORMALIZATION_METHOD,
|
||||
AudioPreparationError,
|
||||
prepare_audio,
|
||||
)
|
||||
|
||||
|
||||
def _write_wav(
|
||||
path: Path, *, channels: int = 1, sample_rate: int = 16_000, sample_width: int = 2
|
||||
) -> None:
|
||||
with wave.open(str(path), "wb") as recording:
|
||||
recording.setnchannels(channels)
|
||||
recording.setsampwidth(sample_width)
|
||||
recording.setframerate(sample_rate)
|
||||
recording.writeframes(b"\x00" * channels * sample_width * 32)
|
||||
|
||||
|
||||
def _successful_runner(commands: list[list[str]]):
|
||||
def run(command, **kwargs):
|
||||
commands.append(list(command))
|
||||
_write_wav(Path(command[-1]))
|
||||
return subprocess.CompletedProcess(command, 0, "", "")
|
||||
|
||||
return run
|
||||
|
||||
|
||||
class AudioPreparationTests(unittest.TestCase):
|
||||
def test_supported_inputs_are_prepared_with_normalization_on_and_off(self) -> None:
|
||||
for suffix in (".wav", ".flac", ".m4a"):
|
||||
for normalization_enabled in (True, False):
|
||||
with (
|
||||
self.subTest(
|
||||
suffix=suffix, normalization_enabled=normalization_enabled
|
||||
),
|
||||
tempfile.TemporaryDirectory() as directory,
|
||||
):
|
||||
root = Path(directory)
|
||||
source = root / f"meeting{suffix}"
|
||||
if suffix == ".wav":
|
||||
_write_wav(source)
|
||||
else:
|
||||
source.write_bytes(b"original encoded audio")
|
||||
original = source.read_bytes()
|
||||
destination = root / "run" / "audio" / "prepared.wav"
|
||||
commands: list[list[str]] = []
|
||||
|
||||
with patch(
|
||||
"src.meeting_lab.audio.preparation.shutil.which",
|
||||
return_value="/usr/bin/ffmpeg",
|
||||
):
|
||||
result = prepare_audio(
|
||||
source,
|
||||
destination,
|
||||
normalization_enabled=normalization_enabled,
|
||||
runner=_successful_runner(commands),
|
||||
)
|
||||
|
||||
self.assertEqual(source.read_bytes(), original)
|
||||
self.assertEqual(result.prepared_path, destination)
|
||||
with wave.open(str(destination), "rb") as recording:
|
||||
self.assertEqual(recording.getnchannels(), 1)
|
||||
self.assertEqual(recording.getframerate(), 16_000)
|
||||
self.assertEqual(recording.getsampwidth(), 2)
|
||||
self.assertEqual(recording.getcomptype(), "NONE")
|
||||
self.assertEqual(commands[0][commands[0].index("-ac") + 1], "1")
|
||||
self.assertEqual(commands[0][commands[0].index("-ar") + 1], "16000")
|
||||
self.assertEqual(
|
||||
commands[0][commands[0].index("-c:a") + 1], "pcm_s16le"
|
||||
)
|
||||
self.assertEqual("-af" in commands[0], normalization_enabled)
|
||||
if normalization_enabled:
|
||||
self.assertEqual(
|
||||
commands[0][commands[0].index("-af") + 1],
|
||||
DEFAULT_NORMALIZATION_FILTER,
|
||||
)
|
||||
self.assertEqual(
|
||||
result.normalization_enabled, normalization_enabled
|
||||
)
|
||||
|
||||
def test_normalization_defaults_to_on_and_explicit_on_matches(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
source = root / "meeting.wav"
|
||||
_write_wav(source)
|
||||
commands: list[list[str]] = []
|
||||
with patch(
|
||||
"src.meeting_lab.audio.preparation.shutil.which",
|
||||
return_value="/usr/bin/ffmpeg",
|
||||
):
|
||||
default = prepare_audio(
|
||||
source, root / "default.wav", runner=_successful_runner(commands)
|
||||
)
|
||||
explicit = prepare_audio(
|
||||
source,
|
||||
root / "explicit.wav",
|
||||
normalization_enabled=True,
|
||||
runner=_successful_runner(commands),
|
||||
)
|
||||
|
||||
self.assertTrue(default.normalization_enabled)
|
||||
self.assertTrue(explicit.normalization_enabled)
|
||||
self.assertEqual(
|
||||
commands[0][commands[0].index("-af") + 1],
|
||||
commands[1][commands[1].index("-af") + 1],
|
||||
)
|
||||
|
||||
def test_noncanonical_wav_is_normalized(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
source = root / "stereo-48k.wav"
|
||||
_write_wav(source, channels=2, sample_rate=48_000)
|
||||
destination = root / "prepared.wav"
|
||||
|
||||
with patch(
|
||||
"src.meeting_lab.audio.preparation.shutil.which",
|
||||
return_value="/usr/bin/ffmpeg",
|
||||
):
|
||||
prepare_audio(source, destination, runner=_successful_runner([]))
|
||||
|
||||
with wave.open(str(destination), "rb") as recording:
|
||||
self.assertEqual(
|
||||
(recording.getnchannels(), recording.getframerate()), (1, 16_000)
|
||||
)
|
||||
|
||||
def test_ffmpeg_missing_has_actionable_error(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
source = root / "meeting.flac"
|
||||
source.write_bytes(b"audio")
|
||||
|
||||
with (
|
||||
patch(
|
||||
"src.meeting_lab.audio.preparation.shutil.which", return_value=None
|
||||
),
|
||||
self.assertRaisesRegex(AudioPreparationError, "not found on PATH"),
|
||||
):
|
||||
prepare_audio(source, root / "prepared.wav")
|
||||
|
||||
def test_ffmpeg_failure_includes_diagnostic_and_preserves_source(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
source = root / "meeting.m4a"
|
||||
source.write_bytes(b"original")
|
||||
|
||||
def fail(command, **kwargs):
|
||||
return subprocess.CompletedProcess(command, 1, "", "decoder exploded")
|
||||
|
||||
with (
|
||||
patch(
|
||||
"src.meeting_lab.audio.preparation.shutil.which",
|
||||
return_value="/usr/bin/ffmpeg",
|
||||
),
|
||||
self.assertRaisesRegex(AudioPreparationError, "decoder exploded"),
|
||||
):
|
||||
prepare_audio(source, root / "prepared.wav", runner=fail)
|
||||
|
||||
self.assertEqual(source.read_bytes(), b"original")
|
||||
self.assertFalse((root / "prepared.wav").exists())
|
||||
|
||||
def test_prepared_audio_metadata_is_traceable(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
source = root / "unknown_meeting.flac"
|
||||
source.write_bytes(b"source")
|
||||
destination = root / "audio" / "prepared.wav"
|
||||
with patch(
|
||||
"src.meeting_lab.audio.preparation.shutil.which",
|
||||
return_value="/usr/bin/ffmpeg",
|
||||
):
|
||||
result = prepare_audio(
|
||||
source, destination, runner=_successful_runner([])
|
||||
)
|
||||
|
||||
metadata = result.metadata()
|
||||
self.assertEqual(metadata["original_source_name"], "unknown_meeting.flac")
|
||||
self.assertEqual(metadata["original_format"], "flac")
|
||||
self.assertEqual(
|
||||
metadata["prepared_audio_path"], str(destination.resolve())
|
||||
)
|
||||
self.assertEqual(metadata["preparation_method"], "ffmpeg")
|
||||
self.assertTrue(metadata["normalization_enabled"])
|
||||
self.assertEqual(
|
||||
metadata["normalization_method"], DEFAULT_NORMALIZATION_METHOD
|
||||
)
|
||||
self.assertEqual(
|
||||
metadata["normalization_filter"], DEFAULT_NORMALIZATION_FILTER
|
||||
)
|
||||
self.assertEqual(
|
||||
metadata["canonical_output"],
|
||||
{
|
||||
"container": "wav",
|
||||
"codec": "pcm_s16le",
|
||||
"channels": 1,
|
||||
"sample_rate_hz": 16_000,
|
||||
"bits_per_sample": 16,
|
||||
},
|
||||
)
|
||||
|
||||
def test_disabled_normalization_metadata_has_no_method_or_filter(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
source = root / "meeting.m4a"
|
||||
source.write_bytes(b"source")
|
||||
with patch(
|
||||
"src.meeting_lab.audio.preparation.shutil.which",
|
||||
return_value="/usr/bin/ffmpeg",
|
||||
):
|
||||
result = prepare_audio(
|
||||
source,
|
||||
root / "prepared.wav",
|
||||
normalization_enabled=False,
|
||||
runner=_successful_runner([]),
|
||||
)
|
||||
|
||||
metadata = result.metadata()
|
||||
self.assertFalse(metadata["normalization_enabled"])
|
||||
self.assertIsNone(metadata["normalization_method"])
|
||||
self.assertIsNone(metadata["normalization_filter"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -72,6 +72,38 @@ class MeetingContextTests(unittest.TestCase):
|
||||
with self.assertRaisesRegex(MeetingContextValidationError, "invalid value"):
|
||||
validate_meeting_context(data)
|
||||
|
||||
def test_missing_participant_attendance_defaults_to_present(self) -> None:
|
||||
data = copy.deepcopy(self.context.data)
|
||||
del data["participants"][0]["attendance_status"]
|
||||
|
||||
context = create_meeting_context(data)
|
||||
|
||||
self.assertEqual(
|
||||
context.data["participants"][0]["attendance_status"], "present"
|
||||
)
|
||||
|
||||
def test_explicit_mentioned_only_is_preserved(self) -> None:
|
||||
data = copy.deepcopy(self.context.data)
|
||||
data["mentioned_people"][0]["attendance_status"] = "mentioned_only"
|
||||
|
||||
context = create_meeting_context(data)
|
||||
|
||||
self.assertEqual(
|
||||
context.data["mentioned_people"][0]["attendance_status"],
|
||||
"mentioned_only",
|
||||
)
|
||||
self.assertIn(
|
||||
"Mentioned but absent people:", render_meeting_context_for_prompt(context)
|
||||
)
|
||||
|
||||
def test_mentioned_only_person_cannot_be_a_diarized_speaker(self) -> None:
|
||||
data = copy.deepcopy(self.context.data)
|
||||
mentioned_id = data["mentioned_people"][0]["person_id"]
|
||||
data["speaker_mappings"] = {"SPEAKER_00": mentioned_id}
|
||||
|
||||
with self.assertRaisesRegex(MeetingContextValidationError, "unknown participant"):
|
||||
validate_meeting_context(data)
|
||||
|
||||
def test_prompt_representation_is_deterministic(self) -> None:
|
||||
first = render_meeting_context_for_prompt(self.context)
|
||||
second = render_meeting_context_for_prompt(self.context)
|
||||
|
||||
+75
-1
@@ -1,11 +1,14 @@
|
||||
import json
|
||||
import shutil
|
||||
import subprocess
|
||||
import tempfile
|
||||
import unittest
|
||||
from dataclasses import replace
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
from scripts import run_mvp_meeting as cli
|
||||
from src.meeting_lab.audio import PreparedAudio
|
||||
from src.meeting_lab.models.meeting_context import load_meeting_context
|
||||
from src.meeting_lab.orchestration import mvp as mvp_api
|
||||
from src.meeting_lab.orchestration.mvp import MvpMeetingConfig, MvpRunResult
|
||||
@@ -71,6 +74,12 @@ def fake_transcribe(audio, model, output, language, **kwargs):
|
||||
return TranscriptionResult(output, raw, transcript, text, metadata, 0.1)
|
||||
|
||||
|
||||
def fake_prepare(source, destination, **kwargs):
|
||||
destination.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copyfile(source, destination)
|
||||
return PreparedAudio(source, source.suffix.removeprefix("."), destination, "ffmpeg", "ffmpeg")
|
||||
|
||||
|
||||
def fake_protocol(transcript, context, **kwargs):
|
||||
rendered_context = load_meeting_context(context)
|
||||
assert rendered_context.meeting_id == "programmatic-test"
|
||||
@@ -103,6 +112,7 @@ class MvpApiTests(unittest.TestCase):
|
||||
events = []
|
||||
with (
|
||||
patch.object(mvp_api, "transcribe_audio", side_effect=fake_transcribe),
|
||||
patch.object(mvp_api, "prepare_audio", side_effect=fake_prepare),
|
||||
patch.object(
|
||||
mvp_api, "generate_direct_protocol", side_effect=fake_protocol
|
||||
),
|
||||
@@ -142,7 +152,7 @@ class MvpApiTests(unittest.TestCase):
|
||||
mvp_api,
|
||||
"transcribe_audio",
|
||||
side_effect=TranscriptionError("stopped"),
|
||||
):
|
||||
), patch.object(mvp_api, "prepare_audio", side_effect=fake_prepare):
|
||||
result = mvp_api.run_mvp_meeting(config, progress_sink=events.append)
|
||||
|
||||
self.assertEqual(result.exit_code, 2)
|
||||
@@ -168,8 +178,72 @@ class MvpApiTests(unittest.TestCase):
|
||||
self.assertEqual(delegated.diarization, "off")
|
||||
self.assertEqual(delegated.language, "de")
|
||||
self.assertEqual(delegated.whisper_executable, "whisper-cli")
|
||||
self.assertEqual(delegated.ffmpeg_executable, "ffmpeg")
|
||||
self.assertTrue(delegated.audio_normalization)
|
||||
self.assertEqual(api.call_args.kwargs["meeting_context"], context_data())
|
||||
|
||||
def test_cli_explicit_audio_normalization_values_are_propagated(self):
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
config = self.config(root)
|
||||
enabled = cli.config_from_args(
|
||||
cli.parse_args(
|
||||
[
|
||||
str(config.audio_file),
|
||||
"--whisper-model",
|
||||
str(config.whisper_model),
|
||||
"--audio-normalization",
|
||||
]
|
||||
)
|
||||
)
|
||||
disabled = cli.config_from_args(
|
||||
cli.parse_args(
|
||||
[
|
||||
str(config.audio_file),
|
||||
"--whisper-model",
|
||||
str(config.whisper_model),
|
||||
"--no-audio-normalization",
|
||||
]
|
||||
)
|
||||
)
|
||||
|
||||
self.assertTrue(enabled.audio_normalization)
|
||||
self.assertFalse(disabled.audio_normalization)
|
||||
|
||||
def test_transcription_receives_prepared_wav_for_encoded_inputs(self):
|
||||
for suffix in (".flac", ".m4a"):
|
||||
with self.subTest(suffix=suffix), tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
config = self.config(root)
|
||||
encoded = config.audio_file.with_suffix(suffix)
|
||||
config.audio_file.rename(encoded)
|
||||
config = replace(config, audio_file=encoded)
|
||||
received = []
|
||||
|
||||
def capture_transcribe(audio, *args, received_paths=received, **kwargs):
|
||||
received_paths.append(audio)
|
||||
return fake_transcribe(audio, *args, **kwargs)
|
||||
|
||||
with (
|
||||
patch.object(mvp_api, "prepare_audio", side_effect=fake_prepare),
|
||||
patch.object(mvp_api, "transcribe_audio", side_effect=capture_transcribe),
|
||||
patch.object(mvp_api, "generate_direct_protocol", side_effect=fake_protocol),
|
||||
):
|
||||
result = mvp_api.run_mvp_meeting(
|
||||
config, meeting_context=context_data()
|
||||
)
|
||||
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
self.assertEqual(received, [result.run_dir / "audio" / "prepared.wav"])
|
||||
manifest = json.loads(
|
||||
(result.run_dir / "audio" / "input_manifest.json").read_text()
|
||||
)
|
||||
self.assertEqual(manifest["format"], suffix.removeprefix("."))
|
||||
self.assertEqual(
|
||||
manifest["prepared_audio"]["prepared_audio_path"],
|
||||
str((result.run_dir / "audio" / "prepared.wav").resolve()),
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import json
|
||||
import shutil
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import datetime
|
||||
@@ -6,6 +7,11 @@ from pathlib import Path
|
||||
from unittest.mock import Mock, patch
|
||||
|
||||
from scripts import run_mvp_meeting
|
||||
from src.meeting_lab.audio import PreparedAudio
|
||||
from src.meeting_lab.audio.preparation import (
|
||||
DEFAULT_NORMALIZATION_FILTER,
|
||||
DEFAULT_NORMALIZATION_METHOD,
|
||||
)
|
||||
from src.meeting_lab.orchestration import mvp as mvp_api
|
||||
from src.meeting_lab.protocol.generate_direct_protocol import DirectProtocolResult
|
||||
from src.meeting_lab.diarization.backend import DiarizationResult
|
||||
@@ -106,7 +112,28 @@ def fake_diarize(audio_path, output_dir, device_mode, **kwargs):
|
||||
)
|
||||
|
||||
|
||||
def fake_prepare(source, destination, **kwargs):
|
||||
destination.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copyfile(source, destination)
|
||||
normalization_enabled = kwargs.get("normalization_enabled", True)
|
||||
return PreparedAudio(
|
||||
source,
|
||||
source.suffix.removeprefix("."),
|
||||
destination,
|
||||
"ffmpeg",
|
||||
kwargs.get("ffmpeg_executable", "ffmpeg"),
|
||||
normalization_enabled,
|
||||
DEFAULT_NORMALIZATION_METHOD if normalization_enabled else None,
|
||||
DEFAULT_NORMALIZATION_FILTER if normalization_enabled else None,
|
||||
)
|
||||
|
||||
|
||||
class MvpOrchestratorTests(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
patcher = patch.object(mvp_api, "prepare_audio", side_effect=fake_prepare)
|
||||
self.prepare_audio = patcher.start()
|
||||
self.addCleanup(patcher.stop)
|
||||
|
||||
def create_inputs(self, root: Path) -> tuple[Path, Path, Path]:
|
||||
audio = root / "team meeting.wav"
|
||||
whisper_model = root / "ggml-model.bin"
|
||||
@@ -151,6 +178,8 @@ class MvpOrchestratorTests(unittest.TestCase):
|
||||
expected = {
|
||||
"run_metadata.json",
|
||||
"audio/input_manifest.json",
|
||||
"audio/prepared.wav",
|
||||
"audio/preparation_metadata.json",
|
||||
"transcript/whisper_raw.json",
|
||||
"transcript/transcript.json",
|
||||
"transcript/transcript.txt",
|
||||
@@ -167,14 +196,60 @@ class MvpOrchestratorTests(unittest.TestCase):
|
||||
self.assertEqual(metadata["model"], "chosen:model")
|
||||
self.assertIsNone(metadata["failure"])
|
||||
self.assertEqual(whisper.call_count, 1)
|
||||
self.assertEqual(
|
||||
whisper.call_args.args[0], run_dir / "audio" / "prepared.wav"
|
||||
)
|
||||
self.assertEqual(protocol.call_count, 1)
|
||||
self.assertTrue(
|
||||
self.prepare_audio.call_args.kwargs["normalization_enabled"]
|
||||
)
|
||||
preparation = json.loads(
|
||||
(run_dir / "audio" / "preparation_metadata.json").read_text()
|
||||
)
|
||||
self.assertTrue(preparation["normalization_enabled"])
|
||||
|
||||
def test_cli_can_disable_audio_normalization_without_bypassing_preparation(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
args = self.args(root, ["--no-audio-normalization"])
|
||||
with (
|
||||
patch.object(mvp_api, "transcribe_audio", side_effect=fake_transcribe) as whisper,
|
||||
patch.object(
|
||||
mvp_api,
|
||||
"generate_direct_protocol",
|
||||
return_value=protocol_result(),
|
||||
),
|
||||
):
|
||||
code, run_dir, _ = run_mvp_meeting.run(args)
|
||||
|
||||
self.assertEqual(code, 0)
|
||||
self.prepare_audio.assert_called_once()
|
||||
self.assertFalse(
|
||||
self.prepare_audio.call_args.kwargs["normalization_enabled"]
|
||||
)
|
||||
self.assertEqual(
|
||||
whisper.call_args.args[0], run_dir / "audio" / "prepared.wav"
|
||||
)
|
||||
preparation = json.loads(
|
||||
(run_dir / "audio" / "preparation_metadata.json").read_text()
|
||||
)
|
||||
self.assertFalse(preparation["normalization_enabled"])
|
||||
self.assertIsNone(preparation["normalization_method"])
|
||||
self.assertIsNone(preparation["normalization_filter"])
|
||||
|
||||
def test_context_model_endpoint_and_whisper_options_are_forwarded(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
args = self.args(
|
||||
root,
|
||||
["--whisper-executable", "/tools/whisper-cli", "--threads", "4"],
|
||||
[
|
||||
"--whisper-executable",
|
||||
"/tools/whisper-cli",
|
||||
"--ffmpeg-executable",
|
||||
"/tools/ffmpeg",
|
||||
"--threads",
|
||||
"4",
|
||||
],
|
||||
)
|
||||
with (
|
||||
patch.object(mvp_api, "transcribe_audio", side_effect=fake_transcribe) as whisper,
|
||||
@@ -190,6 +265,10 @@ class MvpOrchestratorTests(unittest.TestCase):
|
||||
self.assertEqual(whisper.call_args.args[3], "de")
|
||||
self.assertEqual(whisper.call_args.kwargs["executable"], "/tools/whisper-cli")
|
||||
self.assertEqual(whisper.call_args.kwargs["threads"], "4")
|
||||
self.assertEqual(
|
||||
self.prepare_audio.call_args.kwargs["ffmpeg_executable"],
|
||||
"/tools/ffmpeg",
|
||||
)
|
||||
self.assertEqual(protocol.call_args.args[1], run_dir / "context/meeting_context.yaml")
|
||||
self.assertEqual(protocol.call_args.kwargs["model"], "chosen:model")
|
||||
self.assertEqual(
|
||||
@@ -245,6 +324,9 @@ class MvpOrchestratorTests(unittest.TestCase):
|
||||
|
||||
self.assertEqual(code, 0)
|
||||
self.assertEqual(diarization.call_args.args[2], "gpu")
|
||||
self.assertEqual(
|
||||
diarization.call_args.args[0], run_dir / "audio" / "prepared.wav"
|
||||
)
|
||||
self.assertEqual(diarization.call_args.kwargs["runtime"], "container")
|
||||
self.assertEqual(
|
||||
diarization.call_args.kwargs["container_args"], ("--device=/dev/kfd",)
|
||||
|
||||
Reference in New Issue
Block a user