Record glossary provenance without transcript mutation
This commit is contained in:
@@ -299,3 +299,7 @@ protocol.
|
||||
Protocol calls accept an optional positive `protocol_num_thread` setting through
|
||||
both initial processing and regeneration. With `None`, Ollama receives no
|
||||
`num_thread` override and selects its own thread configuration.
|
||||
|
||||
Configured glossary aliases are diagnostic metadata only. Meeting Context supplies
|
||||
terminology guidance; direct-protocol transcript text is never alias-substituted.
|
||||
See `docs/protocol-generation-regression.md` for the frozen-input regression.
|
||||
|
||||
@@ -60,3 +60,8 @@ also exceeds the budget, generation fails before model lookup or generation;
|
||||
it never truncates, chunks, summarizes, retries, or makes multiple protocol
|
||||
calls implicitly. Full diarization artifacts are never overwritten by this
|
||||
selection.
|
||||
|
||||
Glossary aliases are recorded as configuration provenance, never applied as
|
||||
deterministic replacements to the compact/plain protocol input. Canonical
|
||||
terminology guides generation through Meeting Context. Raw Whisper and
|
||||
diarization artifacts remain unchanged. `glossary_replacements` is always empty.
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
# Direct-protocol regression reproduction
|
||||
|
||||
The August 2026 glossary regression was isolated with frozen inputs. Condition
|
||||
A used the original derived transcript; condition B differed only by these
|
||||
seven deterministic substitutions:
|
||||
|
||||
- `Carbofool` to `Carbofol` (two occurrences)
|
||||
- `Bento Fix` to `Bentofix` (two occurrences)
|
||||
- `Sikirgut-Heistlöse` to `Secugrid HS` (one occurrence)
|
||||
- `Lumini` to `Luminy` (two occurrences)
|
||||
|
||||
To repeat the manual comparison, copy the investigated run to a new temporary
|
||||
directory, retain its Meeting Context and speaker mapping, and invoke
|
||||
`regenerate_mvp_protocol` through the same parameters used by Meeting
|
||||
Assistant. Never run the comparison in the historical run directory. Preserve
|
||||
the model, `num_ctx`, `num_predict`, temperature, think setting, thread setting,
|
||||
and Meeting Context. Compare the new generation's `exact_prompt.txt` and
|
||||
`transcript_input.txt` with the frozen A and B artifacts before comparing model
|
||||
output.
|
||||
|
||||
The required fixed condition is A: configured glossary aliases remain visible
|
||||
in Meeting Context terminology guidance, while `transcript_input.txt` retains
|
||||
the original seven source spellings. Automated tests mock the model and enforce
|
||||
that invariant; this full historical experiment remains an explicit manual LLM
|
||||
validation so normal tests do not depend on a local model.
|
||||
@@ -56,6 +56,7 @@ class MvpMeetingConfig:
|
||||
threads: str | int = "auto"
|
||||
model: str = DEFAULT_MODEL
|
||||
ollama_endpoint: str = DEFAULT_ENDPOINT
|
||||
glossary_aliases: Mapping[str, str] | None = None
|
||||
protocol_num_thread: int | None = None
|
||||
protocol_num_ctx: int = DEFAULT_NUM_CTX
|
||||
protocol_safe_input_token_budget: int = DEFAULT_SAFE_INPUT_TOKEN_BUDGET
|
||||
@@ -78,6 +79,7 @@ def regenerate_mvp_protocol(
|
||||
meeting_context: ContextInput,
|
||||
model: str = DEFAULT_MODEL,
|
||||
ollama_endpoint: str = DEFAULT_ENDPOINT,
|
||||
glossary_aliases: Mapping[str, str] | None = None,
|
||||
protocol_num_thread: int | None = None,
|
||||
protocol_num_ctx: int = DEFAULT_NUM_CTX,
|
||||
protocol_safe_input_token_budget: int = DEFAULT_SAFE_INPUT_TOKEN_BUDGET,
|
||||
@@ -119,6 +121,7 @@ def regenerate_mvp_protocol(
|
||||
endpoint=ollama_endpoint,
|
||||
num_ctx=protocol_num_ctx,
|
||||
num_thread=protocol_num_thread,
|
||||
glossary_aliases=glossary_aliases,
|
||||
safe_input_token_budget=protocol_safe_input_token_budget,
|
||||
)
|
||||
protocol_path = _persist_protocol(run_dir, result)
|
||||
@@ -438,6 +441,7 @@ def run_mvp_meeting(
|
||||
endpoint=config.ollama_endpoint,
|
||||
num_ctx=config.protocol_num_ctx,
|
||||
num_thread=config.protocol_num_thread,
|
||||
glossary_aliases=config.glossary_aliases,
|
||||
safe_input_token_budget=config.protocol_safe_input_token_budget,
|
||||
)
|
||||
stage_runtimes["protocol"] = round(time.perf_counter() - stage_started, 3)
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable
|
||||
@@ -174,6 +175,7 @@ def generate_direct_protocol(
|
||||
safe_input_token_budget: int = DEFAULT_SAFE_INPUT_TOKEN_BUDGET,
|
||||
model_check: Callable[[str, str, int], dict[str, Any]] = require_model,
|
||||
generation_call: Callable[..., OllamaGeneration] = generate_once,
|
||||
glossary_aliases: Mapping[str, str] | None = None,
|
||||
) -> DirectProtocolResult:
|
||||
transcript = _load_transcript_document(transcript_path)
|
||||
context: MeetingContext | None = (
|
||||
@@ -238,6 +240,8 @@ def generate_direct_protocol(
|
||||
else None
|
||||
),
|
||||
"speaker_mapping_count": len(context.speaker_mappings) if context else 0,
|
||||
"glossary_aliases_configured": dict(sorted((glossary_aliases or {}).items())),
|
||||
"glossary_replacements": [],
|
||||
}
|
||||
return DirectProtocolResult(
|
||||
protocol_text=generation.text,
|
||||
|
||||
@@ -9,6 +9,8 @@ from unittest.mock import patch
|
||||
|
||||
from scripts import run_mvp_meeting as cli
|
||||
from src.meeting_lab.audio import PreparedAudio
|
||||
from src.meeting_lab.llm.ollama import OllamaGeneration
|
||||
from src.meeting_lab.protocol.generate_direct_protocol import generate_direct_protocol
|
||||
from src.meeting_lab.models.meeting_context import load_meeting_context
|
||||
from src.meeting_lab.orchestration import mvp as mvp_api
|
||||
from src.meeting_lab.orchestration.mvp import MvpMeetingConfig, MvpRunResult
|
||||
@@ -202,6 +204,85 @@ class MvpApiTests(unittest.TestCase):
|
||||
persisted = load_meeting_context(run_dir / "context/meeting_context.yaml")
|
||||
self.assertEqual(persisted.speaker_mappings, {"SPEAKER_00": "person-1"})
|
||||
|
||||
def test_regeneration_keeps_glossary_out_of_transcript_and_in_context(self):
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
run_dir = Path(directory)
|
||||
diarization_dir = run_dir / "diarization"
|
||||
diarization_dir.mkdir()
|
||||
source = diarization_dir / "transcript_diarized.json"
|
||||
source.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"text": "SPEAKER_00: Lumini.",
|
||||
"segments": [
|
||||
{
|
||||
"start": 0.0,
|
||||
"end": 1.0,
|
||||
"speaker_id": "SPEAKER_00",
|
||||
"text": "Lumini.",
|
||||
}
|
||||
],
|
||||
"speaker_labels_anonymous": True,
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
source_before = source.read_bytes()
|
||||
transcript_dir = run_dir / "transcript"
|
||||
transcript_dir.mkdir()
|
||||
whisper_source = transcript_dir / "transcript.json"
|
||||
whisper_source.write_text(
|
||||
json.dumps({"text": "Lumini.", "segments": []}), encoding="utf-8"
|
||||
)
|
||||
whisper_source_before = whisper_source.read_bytes()
|
||||
context = context_data()
|
||||
context["known_entities"] = {
|
||||
"Authoritative terminology": ["Luminy (aliases: Lumini)"]
|
||||
}
|
||||
|
||||
def generate_without_network(transcript, context_path, **kwargs):
|
||||
return generate_direct_protocol(
|
||||
transcript,
|
||||
context_path,
|
||||
model=kwargs["model"],
|
||||
num_ctx=kwargs["num_ctx"],
|
||||
num_thread=kwargs["num_thread"],
|
||||
safe_input_token_budget=kwargs["safe_input_token_budget"],
|
||||
glossary_aliases=kwargs["glossary_aliases"],
|
||||
model_check=lambda *_: {},
|
||||
generation_call=lambda *_args, **_kwargs: OllamaGeneration(
|
||||
raw_response={"response": "# Protocol", "done": True},
|
||||
text="# Protocol",
|
||||
client_wall_time_seconds=0.1,
|
||||
),
|
||||
)
|
||||
|
||||
with (
|
||||
patch.object(
|
||||
mvp_api,
|
||||
"generate_direct_protocol",
|
||||
side_effect=generate_without_network,
|
||||
),
|
||||
):
|
||||
mvp_api.regenerate_mvp_protocol(
|
||||
run_dir,
|
||||
meeting_context=context,
|
||||
glossary_aliases={"Lumini": "Luminy"},
|
||||
)
|
||||
|
||||
generation = run_dir / "protocol"
|
||||
transcript_input = (generation / "transcript_input.txt").read_text()
|
||||
exact_prompt = (generation / "exact_prompt.txt").read_text()
|
||||
metadata = json.loads((generation / "runtime_metadata.json").read_text())
|
||||
self.assertIn("SPEAKER_00: Lumini.", transcript_input)
|
||||
self.assertNotIn("SPEAKER_00: Luminy.", transcript_input)
|
||||
self.assertIn("Luminy (aliases: Lumini)", exact_prompt)
|
||||
self.assertIn("SPEAKER_00", exact_prompt)
|
||||
self.assertIn("Test Person", exact_prompt)
|
||||
self.assertEqual(metadata["glossary_replacements"], [])
|
||||
self.assertEqual(source.read_bytes(), source_before)
|
||||
self.assertEqual(whisper_source.read_bytes(), whisper_source_before)
|
||||
|
||||
def test_failure_emits_terminal_failure_event(self):
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
|
||||
@@ -137,6 +137,43 @@ class ProtocolInputBudgetTests(unittest.TestCase):
|
||||
self.assertEqual(result.transcript_input, compact)
|
||||
self.assertEqual(call.call_count, 1)
|
||||
|
||||
def test_glossary_aliases_do_not_mutate_protocol_input_or_source_artifact(
|
||||
self,
|
||||
) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
document = {
|
||||
"text": "Lumini und Carbofool.",
|
||||
"segments": [
|
||||
{
|
||||
"start": 0.0,
|
||||
"end": 1.0,
|
||||
"speaker_id": "SPEAKER_00",
|
||||
"text": "Lumini und Carbofool.",
|
||||
}
|
||||
],
|
||||
"speaker_labels_anonymous": True,
|
||||
}
|
||||
transcript = self._write(root, document)
|
||||
source_before = transcript.read_bytes()
|
||||
|
||||
result = generate_direct_protocol(
|
||||
transcript,
|
||||
glossary_aliases={"Lumini": "Luminy", "Carbofool": "Carbofol"},
|
||||
model_check=Mock(return_value={}),
|
||||
generation_call=Mock(return_value=completed_generation()),
|
||||
)
|
||||
|
||||
self.assertEqual(transcript.read_bytes(), source_before)
|
||||
|
||||
self.assertIn("SPEAKER_00: Lumini und Carbofool.", result.transcript_input)
|
||||
self.assertIn("SPEAKER_00: Lumini und Carbofool.", result.exact_prompt)
|
||||
self.assertEqual(result.runtime_metadata["glossary_replacements"], [])
|
||||
self.assertEqual(
|
||||
result.runtime_metadata["glossary_aliases_configured"],
|
||||
{"Carbofool": "Carbofol", "Lumini": "Luminy"},
|
||||
)
|
||||
|
||||
def test_mapped_speakers_and_statements_reach_final_ollama_payload(self) -> None:
|
||||
diarized = {
|
||||
"text": "",
|
||||
|
||||
Reference in New Issue
Block a user