Record glossary provenance without transcript mutation
This commit is contained in:
@@ -299,3 +299,7 @@ protocol.
|
|||||||
Protocol calls accept an optional positive `protocol_num_thread` setting through
|
Protocol calls accept an optional positive `protocol_num_thread` setting through
|
||||||
both initial processing and regeneration. With `None`, Ollama receives no
|
both initial processing and regeneration. With `None`, Ollama receives no
|
||||||
`num_thread` override and selects its own thread configuration.
|
`num_thread` override and selects its own thread configuration.
|
||||||
|
|
||||||
|
Configured glossary aliases are diagnostic metadata only. Meeting Context supplies
|
||||||
|
terminology guidance; direct-protocol transcript text is never alias-substituted.
|
||||||
|
See `docs/protocol-generation-regression.md` for the frozen-input regression.
|
||||||
|
|||||||
@@ -60,3 +60,8 @@ also exceeds the budget, generation fails before model lookup or generation;
|
|||||||
it never truncates, chunks, summarizes, retries, or makes multiple protocol
|
it never truncates, chunks, summarizes, retries, or makes multiple protocol
|
||||||
calls implicitly. Full diarization artifacts are never overwritten by this
|
calls implicitly. Full diarization artifacts are never overwritten by this
|
||||||
selection.
|
selection.
|
||||||
|
|
||||||
|
Glossary aliases are recorded as configuration provenance, never applied as
|
||||||
|
deterministic replacements to the compact/plain protocol input. Canonical
|
||||||
|
terminology guides generation through Meeting Context. Raw Whisper and
|
||||||
|
diarization artifacts remain unchanged. `glossary_replacements` is always empty.
|
||||||
|
|||||||
@@ -0,0 +1,25 @@
|
|||||||
|
# Direct-protocol regression reproduction
|
||||||
|
|
||||||
|
The August 2026 glossary regression was isolated with frozen inputs. Condition
|
||||||
|
A used the original derived transcript; condition B differed only by these
|
||||||
|
seven deterministic substitutions:
|
||||||
|
|
||||||
|
- `Carbofool` to `Carbofol` (two occurrences)
|
||||||
|
- `Bento Fix` to `Bentofix` (two occurrences)
|
||||||
|
- `Sikirgut-Heistlöse` to `Secugrid HS` (one occurrence)
|
||||||
|
- `Lumini` to `Luminy` (two occurrences)
|
||||||
|
|
||||||
|
To repeat the manual comparison, copy the investigated run to a new temporary
|
||||||
|
directory, retain its Meeting Context and speaker mapping, and invoke
|
||||||
|
`regenerate_mvp_protocol` through the same parameters used by Meeting
|
||||||
|
Assistant. Never run the comparison in the historical run directory. Preserve
|
||||||
|
the model, `num_ctx`, `num_predict`, temperature, think setting, thread setting,
|
||||||
|
and Meeting Context. Compare the new generation's `exact_prompt.txt` and
|
||||||
|
`transcript_input.txt` with the frozen A and B artifacts before comparing model
|
||||||
|
output.
|
||||||
|
|
||||||
|
The required fixed condition is A: configured glossary aliases remain visible
|
||||||
|
in Meeting Context terminology guidance, while `transcript_input.txt` retains
|
||||||
|
the original seven source spellings. Automated tests mock the model and enforce
|
||||||
|
that invariant; this full historical experiment remains an explicit manual LLM
|
||||||
|
validation so normal tests do not depend on a local model.
|
||||||
@@ -56,6 +56,7 @@ class MvpMeetingConfig:
|
|||||||
threads: str | int = "auto"
|
threads: str | int = "auto"
|
||||||
model: str = DEFAULT_MODEL
|
model: str = DEFAULT_MODEL
|
||||||
ollama_endpoint: str = DEFAULT_ENDPOINT
|
ollama_endpoint: str = DEFAULT_ENDPOINT
|
||||||
|
glossary_aliases: Mapping[str, str] | None = None
|
||||||
protocol_num_thread: int | None = None
|
protocol_num_thread: int | None = None
|
||||||
protocol_num_ctx: int = DEFAULT_NUM_CTX
|
protocol_num_ctx: int = DEFAULT_NUM_CTX
|
||||||
protocol_safe_input_token_budget: int = DEFAULT_SAFE_INPUT_TOKEN_BUDGET
|
protocol_safe_input_token_budget: int = DEFAULT_SAFE_INPUT_TOKEN_BUDGET
|
||||||
@@ -78,6 +79,7 @@ def regenerate_mvp_protocol(
|
|||||||
meeting_context: ContextInput,
|
meeting_context: ContextInput,
|
||||||
model: str = DEFAULT_MODEL,
|
model: str = DEFAULT_MODEL,
|
||||||
ollama_endpoint: str = DEFAULT_ENDPOINT,
|
ollama_endpoint: str = DEFAULT_ENDPOINT,
|
||||||
|
glossary_aliases: Mapping[str, str] | None = None,
|
||||||
protocol_num_thread: int | None = None,
|
protocol_num_thread: int | None = None,
|
||||||
protocol_num_ctx: int = DEFAULT_NUM_CTX,
|
protocol_num_ctx: int = DEFAULT_NUM_CTX,
|
||||||
protocol_safe_input_token_budget: int = DEFAULT_SAFE_INPUT_TOKEN_BUDGET,
|
protocol_safe_input_token_budget: int = DEFAULT_SAFE_INPUT_TOKEN_BUDGET,
|
||||||
@@ -119,6 +121,7 @@ def regenerate_mvp_protocol(
|
|||||||
endpoint=ollama_endpoint,
|
endpoint=ollama_endpoint,
|
||||||
num_ctx=protocol_num_ctx,
|
num_ctx=protocol_num_ctx,
|
||||||
num_thread=protocol_num_thread,
|
num_thread=protocol_num_thread,
|
||||||
|
glossary_aliases=glossary_aliases,
|
||||||
safe_input_token_budget=protocol_safe_input_token_budget,
|
safe_input_token_budget=protocol_safe_input_token_budget,
|
||||||
)
|
)
|
||||||
protocol_path = _persist_protocol(run_dir, result)
|
protocol_path = _persist_protocol(run_dir, result)
|
||||||
@@ -438,6 +441,7 @@ def run_mvp_meeting(
|
|||||||
endpoint=config.ollama_endpoint,
|
endpoint=config.ollama_endpoint,
|
||||||
num_ctx=config.protocol_num_ctx,
|
num_ctx=config.protocol_num_ctx,
|
||||||
num_thread=config.protocol_num_thread,
|
num_thread=config.protocol_num_thread,
|
||||||
|
glossary_aliases=config.glossary_aliases,
|
||||||
safe_input_token_budget=config.protocol_safe_input_token_budget,
|
safe_input_token_budget=config.protocol_safe_input_token_budget,
|
||||||
)
|
)
|
||||||
stage_runtimes["protocol"] = round(time.perf_counter() - stage_started, 3)
|
stage_runtimes["protocol"] = round(time.perf_counter() - stage_started, 3)
|
||||||
|
|||||||
@@ -3,6 +3,7 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import json
|
import json
|
||||||
|
from collections.abc import Mapping
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Callable
|
from typing import Any, Callable
|
||||||
@@ -174,6 +175,7 @@ def generate_direct_protocol(
|
|||||||
safe_input_token_budget: int = DEFAULT_SAFE_INPUT_TOKEN_BUDGET,
|
safe_input_token_budget: int = DEFAULT_SAFE_INPUT_TOKEN_BUDGET,
|
||||||
model_check: Callable[[str, str, int], dict[str, Any]] = require_model,
|
model_check: Callable[[str, str, int], dict[str, Any]] = require_model,
|
||||||
generation_call: Callable[..., OllamaGeneration] = generate_once,
|
generation_call: Callable[..., OllamaGeneration] = generate_once,
|
||||||
|
glossary_aliases: Mapping[str, str] | None = None,
|
||||||
) -> DirectProtocolResult:
|
) -> DirectProtocolResult:
|
||||||
transcript = _load_transcript_document(transcript_path)
|
transcript = _load_transcript_document(transcript_path)
|
||||||
context: MeetingContext | None = (
|
context: MeetingContext | None = (
|
||||||
@@ -238,6 +240,8 @@ def generate_direct_protocol(
|
|||||||
else None
|
else None
|
||||||
),
|
),
|
||||||
"speaker_mapping_count": len(context.speaker_mappings) if context else 0,
|
"speaker_mapping_count": len(context.speaker_mappings) if context else 0,
|
||||||
|
"glossary_aliases_configured": dict(sorted((glossary_aliases or {}).items())),
|
||||||
|
"glossary_replacements": [],
|
||||||
}
|
}
|
||||||
return DirectProtocolResult(
|
return DirectProtocolResult(
|
||||||
protocol_text=generation.text,
|
protocol_text=generation.text,
|
||||||
|
|||||||
@@ -9,6 +9,8 @@ from unittest.mock import patch
|
|||||||
|
|
||||||
from scripts import run_mvp_meeting as cli
|
from scripts import run_mvp_meeting as cli
|
||||||
from src.meeting_lab.audio import PreparedAudio
|
from src.meeting_lab.audio import PreparedAudio
|
||||||
|
from src.meeting_lab.llm.ollama import OllamaGeneration
|
||||||
|
from src.meeting_lab.protocol.generate_direct_protocol import generate_direct_protocol
|
||||||
from src.meeting_lab.models.meeting_context import load_meeting_context
|
from src.meeting_lab.models.meeting_context import load_meeting_context
|
||||||
from src.meeting_lab.orchestration import mvp as mvp_api
|
from src.meeting_lab.orchestration import mvp as mvp_api
|
||||||
from src.meeting_lab.orchestration.mvp import MvpMeetingConfig, MvpRunResult
|
from src.meeting_lab.orchestration.mvp import MvpMeetingConfig, MvpRunResult
|
||||||
@@ -202,6 +204,85 @@ class MvpApiTests(unittest.TestCase):
|
|||||||
persisted = load_meeting_context(run_dir / "context/meeting_context.yaml")
|
persisted = load_meeting_context(run_dir / "context/meeting_context.yaml")
|
||||||
self.assertEqual(persisted.speaker_mappings, {"SPEAKER_00": "person-1"})
|
self.assertEqual(persisted.speaker_mappings, {"SPEAKER_00": "person-1"})
|
||||||
|
|
||||||
|
def test_regeneration_keeps_glossary_out_of_transcript_and_in_context(self):
|
||||||
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
|
run_dir = Path(directory)
|
||||||
|
diarization_dir = run_dir / "diarization"
|
||||||
|
diarization_dir.mkdir()
|
||||||
|
source = diarization_dir / "transcript_diarized.json"
|
||||||
|
source.write_text(
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"text": "SPEAKER_00: Lumini.",
|
||||||
|
"segments": [
|
||||||
|
{
|
||||||
|
"start": 0.0,
|
||||||
|
"end": 1.0,
|
||||||
|
"speaker_id": "SPEAKER_00",
|
||||||
|
"text": "Lumini.",
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"speaker_labels_anonymous": True,
|
||||||
|
}
|
||||||
|
),
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
source_before = source.read_bytes()
|
||||||
|
transcript_dir = run_dir / "transcript"
|
||||||
|
transcript_dir.mkdir()
|
||||||
|
whisper_source = transcript_dir / "transcript.json"
|
||||||
|
whisper_source.write_text(
|
||||||
|
json.dumps({"text": "Lumini.", "segments": []}), encoding="utf-8"
|
||||||
|
)
|
||||||
|
whisper_source_before = whisper_source.read_bytes()
|
||||||
|
context = context_data()
|
||||||
|
context["known_entities"] = {
|
||||||
|
"Authoritative terminology": ["Luminy (aliases: Lumini)"]
|
||||||
|
}
|
||||||
|
|
||||||
|
def generate_without_network(transcript, context_path, **kwargs):
|
||||||
|
return generate_direct_protocol(
|
||||||
|
transcript,
|
||||||
|
context_path,
|
||||||
|
model=kwargs["model"],
|
||||||
|
num_ctx=kwargs["num_ctx"],
|
||||||
|
num_thread=kwargs["num_thread"],
|
||||||
|
safe_input_token_budget=kwargs["safe_input_token_budget"],
|
||||||
|
glossary_aliases=kwargs["glossary_aliases"],
|
||||||
|
model_check=lambda *_: {},
|
||||||
|
generation_call=lambda *_args, **_kwargs: OllamaGeneration(
|
||||||
|
raw_response={"response": "# Protocol", "done": True},
|
||||||
|
text="# Protocol",
|
||||||
|
client_wall_time_seconds=0.1,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
with (
|
||||||
|
patch.object(
|
||||||
|
mvp_api,
|
||||||
|
"generate_direct_protocol",
|
||||||
|
side_effect=generate_without_network,
|
||||||
|
),
|
||||||
|
):
|
||||||
|
mvp_api.regenerate_mvp_protocol(
|
||||||
|
run_dir,
|
||||||
|
meeting_context=context,
|
||||||
|
glossary_aliases={"Lumini": "Luminy"},
|
||||||
|
)
|
||||||
|
|
||||||
|
generation = run_dir / "protocol"
|
||||||
|
transcript_input = (generation / "transcript_input.txt").read_text()
|
||||||
|
exact_prompt = (generation / "exact_prompt.txt").read_text()
|
||||||
|
metadata = json.loads((generation / "runtime_metadata.json").read_text())
|
||||||
|
self.assertIn("SPEAKER_00: Lumini.", transcript_input)
|
||||||
|
self.assertNotIn("SPEAKER_00: Luminy.", transcript_input)
|
||||||
|
self.assertIn("Luminy (aliases: Lumini)", exact_prompt)
|
||||||
|
self.assertIn("SPEAKER_00", exact_prompt)
|
||||||
|
self.assertIn("Test Person", exact_prompt)
|
||||||
|
self.assertEqual(metadata["glossary_replacements"], [])
|
||||||
|
self.assertEqual(source.read_bytes(), source_before)
|
||||||
|
self.assertEqual(whisper_source.read_bytes(), whisper_source_before)
|
||||||
|
|
||||||
def test_failure_emits_terminal_failure_event(self):
|
def test_failure_emits_terminal_failure_event(self):
|
||||||
with tempfile.TemporaryDirectory() as directory:
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
root = Path(directory)
|
root = Path(directory)
|
||||||
|
|||||||
@@ -137,6 +137,43 @@ class ProtocolInputBudgetTests(unittest.TestCase):
|
|||||||
self.assertEqual(result.transcript_input, compact)
|
self.assertEqual(result.transcript_input, compact)
|
||||||
self.assertEqual(call.call_count, 1)
|
self.assertEqual(call.call_count, 1)
|
||||||
|
|
||||||
|
def test_glossary_aliases_do_not_mutate_protocol_input_or_source_artifact(
|
||||||
|
self,
|
||||||
|
) -> None:
|
||||||
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
|
root = Path(directory)
|
||||||
|
document = {
|
||||||
|
"text": "Lumini und Carbofool.",
|
||||||
|
"segments": [
|
||||||
|
{
|
||||||
|
"start": 0.0,
|
||||||
|
"end": 1.0,
|
||||||
|
"speaker_id": "SPEAKER_00",
|
||||||
|
"text": "Lumini und Carbofool.",
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"speaker_labels_anonymous": True,
|
||||||
|
}
|
||||||
|
transcript = self._write(root, document)
|
||||||
|
source_before = transcript.read_bytes()
|
||||||
|
|
||||||
|
result = generate_direct_protocol(
|
||||||
|
transcript,
|
||||||
|
glossary_aliases={"Lumini": "Luminy", "Carbofool": "Carbofol"},
|
||||||
|
model_check=Mock(return_value={}),
|
||||||
|
generation_call=Mock(return_value=completed_generation()),
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(transcript.read_bytes(), source_before)
|
||||||
|
|
||||||
|
self.assertIn("SPEAKER_00: Lumini und Carbofool.", result.transcript_input)
|
||||||
|
self.assertIn("SPEAKER_00: Lumini und Carbofool.", result.exact_prompt)
|
||||||
|
self.assertEqual(result.runtime_metadata["glossary_replacements"], [])
|
||||||
|
self.assertEqual(
|
||||||
|
result.runtime_metadata["glossary_aliases_configured"],
|
||||||
|
{"Carbofool": "Carbofol", "Lumini": "Luminy"},
|
||||||
|
)
|
||||||
|
|
||||||
def test_mapped_speakers_and_statements_reach_final_ollama_payload(self) -> None:
|
def test_mapped_speakers_and_statements_reach_final_ollama_payload(self) -> None:
|
||||||
diarized = {
|
diarized = {
|
||||||
"text": "",
|
"text": "",
|
||||||
|
|||||||
Reference in New Issue
Block a user