Record glossary provenance without transcript mutation
This commit is contained in:
@@ -9,6 +9,8 @@ from unittest.mock import patch
|
||||
|
||||
from scripts import run_mvp_meeting as cli
|
||||
from src.meeting_lab.audio import PreparedAudio
|
||||
from src.meeting_lab.llm.ollama import OllamaGeneration
|
||||
from src.meeting_lab.protocol.generate_direct_protocol import generate_direct_protocol
|
||||
from src.meeting_lab.models.meeting_context import load_meeting_context
|
||||
from src.meeting_lab.orchestration import mvp as mvp_api
|
||||
from src.meeting_lab.orchestration.mvp import MvpMeetingConfig, MvpRunResult
|
||||
@@ -202,6 +204,85 @@ class MvpApiTests(unittest.TestCase):
|
||||
persisted = load_meeting_context(run_dir / "context/meeting_context.yaml")
|
||||
self.assertEqual(persisted.speaker_mappings, {"SPEAKER_00": "person-1"})
|
||||
|
||||
def test_regeneration_keeps_glossary_out_of_transcript_and_in_context(self):
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
run_dir = Path(directory)
|
||||
diarization_dir = run_dir / "diarization"
|
||||
diarization_dir.mkdir()
|
||||
source = diarization_dir / "transcript_diarized.json"
|
||||
source.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"text": "SPEAKER_00: Lumini.",
|
||||
"segments": [
|
||||
{
|
||||
"start": 0.0,
|
||||
"end": 1.0,
|
||||
"speaker_id": "SPEAKER_00",
|
||||
"text": "Lumini.",
|
||||
}
|
||||
],
|
||||
"speaker_labels_anonymous": True,
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
source_before = source.read_bytes()
|
||||
transcript_dir = run_dir / "transcript"
|
||||
transcript_dir.mkdir()
|
||||
whisper_source = transcript_dir / "transcript.json"
|
||||
whisper_source.write_text(
|
||||
json.dumps({"text": "Lumini.", "segments": []}), encoding="utf-8"
|
||||
)
|
||||
whisper_source_before = whisper_source.read_bytes()
|
||||
context = context_data()
|
||||
context["known_entities"] = {
|
||||
"Authoritative terminology": ["Luminy (aliases: Lumini)"]
|
||||
}
|
||||
|
||||
def generate_without_network(transcript, context_path, **kwargs):
|
||||
return generate_direct_protocol(
|
||||
transcript,
|
||||
context_path,
|
||||
model=kwargs["model"],
|
||||
num_ctx=kwargs["num_ctx"],
|
||||
num_thread=kwargs["num_thread"],
|
||||
safe_input_token_budget=kwargs["safe_input_token_budget"],
|
||||
glossary_aliases=kwargs["glossary_aliases"],
|
||||
model_check=lambda *_: {},
|
||||
generation_call=lambda *_args, **_kwargs: OllamaGeneration(
|
||||
raw_response={"response": "# Protocol", "done": True},
|
||||
text="# Protocol",
|
||||
client_wall_time_seconds=0.1,
|
||||
),
|
||||
)
|
||||
|
||||
with (
|
||||
patch.object(
|
||||
mvp_api,
|
||||
"generate_direct_protocol",
|
||||
side_effect=generate_without_network,
|
||||
),
|
||||
):
|
||||
mvp_api.regenerate_mvp_protocol(
|
||||
run_dir,
|
||||
meeting_context=context,
|
||||
glossary_aliases={"Lumini": "Luminy"},
|
||||
)
|
||||
|
||||
generation = run_dir / "protocol"
|
||||
transcript_input = (generation / "transcript_input.txt").read_text()
|
||||
exact_prompt = (generation / "exact_prompt.txt").read_text()
|
||||
metadata = json.loads((generation / "runtime_metadata.json").read_text())
|
||||
self.assertIn("SPEAKER_00: Lumini.", transcript_input)
|
||||
self.assertNotIn("SPEAKER_00: Luminy.", transcript_input)
|
||||
self.assertIn("Luminy (aliases: Lumini)", exact_prompt)
|
||||
self.assertIn("SPEAKER_00", exact_prompt)
|
||||
self.assertIn("Test Person", exact_prompt)
|
||||
self.assertEqual(metadata["glossary_replacements"], [])
|
||||
self.assertEqual(source.read_bytes(), source_before)
|
||||
self.assertEqual(whisper_source.read_bytes(), whisper_source_before)
|
||||
|
||||
def test_failure_emits_terminal_failure_event(self):
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
|
||||
Reference in New Issue
Block a user