245 lines
9.3 KiB
Python
245 lines
9.3 KiB
Python
import copy
|
|
import json
|
|
import unittest
|
|
from pathlib import Path
|
|
from unittest.mock import patch
|
|
|
|
from src.meeting_lab.extraction.extract_chunks import (
|
|
EXTRACTION_CATEGORIES,
|
|
build_prompt,
|
|
extract_chunk,
|
|
)
|
|
from src.meeting_lab.models.meeting_context import (
|
|
MeetingContextValidationError,
|
|
create_meeting_context,
|
|
load_meeting_context,
|
|
render_meeting_context_for_prompt,
|
|
serialize_meeting_context_yaml,
|
|
validate_meeting_context,
|
|
)
|
|
|
|
|
|
CONTEXT_PATH = Path("samples/real_live/project_process_meeting/meeting_context.yaml")
|
|
SCRATCH_DIR = Path(".test-tmp")
|
|
|
|
|
|
class MeetingContextTests(unittest.TestCase):
|
|
def setUp(self) -> None:
|
|
self.context = load_meeting_context(CONTEXT_PATH)
|
|
|
|
def tearDown(self) -> None:
|
|
if not SCRATCH_DIR.exists():
|
|
return
|
|
for path in SCRATCH_DIR.glob("chunk_0*_*.txt"):
|
|
path.unlink()
|
|
for path in SCRATCH_DIR.glob("chunk_0*_*.json"):
|
|
path.unlink()
|
|
|
|
def test_valid_context_loading(self) -> None:
|
|
self.assertEqual(self.context.schema_version, "1")
|
|
self.assertEqual(self.context.meeting_id, "2026-07-27-projektprozess")
|
|
self.assertEqual(self.context.data["meeting"]["language"], "de")
|
|
|
|
def test_duplicate_participant_ids_are_invalid(self) -> None:
|
|
data = copy.deepcopy(self.context.data)
|
|
data["participants"][1]["participant_id"] = data["participants"][0][
|
|
"participant_id"
|
|
]
|
|
|
|
with self.assertRaisesRegex(MeetingContextValidationError, "Duplicate"):
|
|
validate_meeting_context(data)
|
|
|
|
def test_invalid_department_references_are_invalid(self) -> None:
|
|
data = copy.deepcopy(self.context.data)
|
|
data["participants"][0]["department_id"] = "unknown"
|
|
|
|
with self.assertRaisesRegex(MeetingContextValidationError, "unknown department"):
|
|
validate_meeting_context(data)
|
|
|
|
def test_participant_and_mentioned_person_id_collision_is_invalid(self) -> None:
|
|
data = copy.deepcopy(self.context.data)
|
|
data["mentioned_people"][0]["person_id"] = data["participants"][0][
|
|
"participant_id"
|
|
]
|
|
|
|
with self.assertRaisesRegex(MeetingContextValidationError, "collide"):
|
|
validate_meeting_context(data)
|
|
|
|
def test_invalid_attendance_status_is_invalid(self) -> None:
|
|
data = copy.deepcopy(self.context.data)
|
|
data["participants"][0]["attendance_status"] = "remote"
|
|
|
|
with self.assertRaisesRegex(MeetingContextValidationError, "invalid value"):
|
|
validate_meeting_context(data)
|
|
|
|
def test_prompt_representation_is_deterministic(self) -> None:
|
|
first = render_meeting_context_for_prompt(self.context)
|
|
second = render_meeting_context_for_prompt(self.context)
|
|
|
|
self.assertEqual(first, second)
|
|
self.assertIn("MEETING CONTEXT V1", first)
|
|
self.assertIn("- Language: de", first)
|
|
|
|
def test_authoritative_rules_appear_in_prompt(self) -> None:
|
|
prompt = build_prompt(
|
|
"chunk_01_normalized.txt",
|
|
"Martin: Wir besprechen Marketing.",
|
|
meeting_context=self.context,
|
|
)
|
|
|
|
self.assertIn("The participant list is authoritative.", prompt)
|
|
self.assertIn("Mentioned people did not attend this meeting.", prompt)
|
|
self.assertIn("Roles and departments must not be inferred or changed.", prompt)
|
|
self.assertIn("Discussion of a department does not establish responsibility.", prompt)
|
|
self.assertIn(
|
|
"An action item may name a responsible person only when assignment or acceptance is explicit",
|
|
prompt,
|
|
)
|
|
self.assertIn("Objections, suggestions and expertise do not establish ownership.", prompt)
|
|
|
|
def test_extraction_behavior_is_unchanged_without_context(self) -> None:
|
|
response = json.dumps(
|
|
{
|
|
"facts": [],
|
|
"decisions": [],
|
|
"todos": [],
|
|
"open_questions": [],
|
|
"positions": [],
|
|
"technical_details": [],
|
|
}
|
|
)
|
|
|
|
SCRATCH_DIR.mkdir(exist_ok=True)
|
|
chunk_path = SCRATCH_DIR / "chunk_01_normalized.txt"
|
|
output_path = SCRATCH_DIR / "chunk_01_extraction.json"
|
|
chunk_path.write_text("Anna: Keine Entscheidung.", encoding="utf-8")
|
|
|
|
captured_prompts = []
|
|
|
|
def fake_call_ollama(**kwargs):
|
|
captured_prompts.append(kwargs["prompt"])
|
|
return response, {}
|
|
|
|
with patch(
|
|
"src.meeting_lab.extraction.extract_chunks.call_ollama",
|
|
side_effect=fake_call_ollama,
|
|
):
|
|
extraction = extract_chunk(
|
|
chunk_path,
|
|
output_path,
|
|
model="test",
|
|
endpoint="http://example.invalid",
|
|
timeout=1,
|
|
temperature=0.0,
|
|
num_predict=None,
|
|
num_ctx=None,
|
|
)
|
|
|
|
self.assertEqual(set(extraction), set(EXTRACTION_CATEGORIES))
|
|
self.assertNotIn("context", extraction)
|
|
self.assertNotIn("MEETING CONTEXT V1", captured_prompts[0])
|
|
|
|
def test_extraction_result_records_meeting_context_provenance(self) -> None:
|
|
response = json.dumps(
|
|
{
|
|
"facts": [],
|
|
"decisions": [],
|
|
"todos": [],
|
|
"open_questions": [],
|
|
"positions": [],
|
|
"technical_details": [],
|
|
}
|
|
)
|
|
|
|
SCRATCH_DIR.mkdir(exist_ok=True)
|
|
chunk_path = SCRATCH_DIR / "chunk_02_normalized.txt"
|
|
output_path = SCRATCH_DIR / "chunk_02_extraction.json"
|
|
chunk_path.write_text("Martin: Hallo.", encoding="utf-8")
|
|
|
|
with patch(
|
|
"src.meeting_lab.extraction.extract_chunks.call_ollama",
|
|
return_value=(response, {}),
|
|
):
|
|
extraction = extract_chunk(
|
|
chunk_path,
|
|
output_path,
|
|
model="test",
|
|
endpoint="http://example.invalid",
|
|
timeout=1,
|
|
temperature=0.0,
|
|
num_predict=None,
|
|
num_ctx=None,
|
|
meeting_context=self.context,
|
|
)
|
|
|
|
written = json.loads(output_path.read_text(encoding="utf-8"))
|
|
|
|
self.assertEqual(
|
|
extraction["context"]["meeting_id"],
|
|
"2026-07-27-projektprozess",
|
|
)
|
|
self.assertEqual(extraction["context"]["schema_version"], "1")
|
|
self.assertEqual(written["context"], extraction["context"])
|
|
self.assertNotIn("participants", written["context"])
|
|
|
|
def test_real_context_keeps_metadata_separate_from_assignment(self) -> None:
|
|
prompt_context = render_meeting_context_for_prompt(self.context)
|
|
|
|
self.assertIn("Björn", prompt_context)
|
|
self.assertIn("department: Marketing", prompt_context)
|
|
self.assertIn("Mentioned but absent people:", prompt_context)
|
|
self.assertIn("Jovana", prompt_context)
|
|
self.assertIn("Discussion of a department does not establish responsibility.", prompt_context)
|
|
self.assertIn("assignment or acceptance is explicit", prompt_context)
|
|
self.assertNotIn("responsible: Björn", prompt_context)
|
|
self.assertNotIn("responsible: Jovana", prompt_context)
|
|
|
|
def test_existing_context_without_speaker_mappings_remains_valid(self) -> None:
|
|
self.assertEqual(self.context.speaker_mappings, {})
|
|
self.assertIsNone(self.context.participant_for_speaker("SPEAKER_00"))
|
|
|
|
def test_explicit_speaker_mapping_is_authoritative(self) -> None:
|
|
data = copy.deepcopy(self.context.data)
|
|
participant = data["participants"][0]
|
|
data["speaker_mappings"] = {"SPEAKER_03": participant["participant_id"]}
|
|
|
|
context = create_meeting_context(data)
|
|
rendered = render_meeting_context_for_prompt(context)
|
|
|
|
self.assertEqual(
|
|
context.participant_for_speaker("SPEAKER_03")["participant_id"],
|
|
participant["participant_id"],
|
|
)
|
|
self.assertIsNone(context.participant_for_speaker("SPEAKER_04"))
|
|
self.assertIn("Confirmed diarization speaker mappings (authoritative)", rendered)
|
|
self.assertIn("Unmapped SPEAKER_XX labels must remain anonymous", rendered)
|
|
|
|
def test_speaker_mapping_must_reference_existing_participant(self) -> None:
|
|
data = copy.deepcopy(self.context.data)
|
|
data["speaker_mappings"] = {"SPEAKER_00": "unknown-person"}
|
|
with self.assertRaisesRegex(MeetingContextValidationError, "unknown participant"):
|
|
validate_meeting_context(data)
|
|
|
|
def test_speaker_mapping_label_must_use_pyannote_shape(self) -> None:
|
|
data = copy.deepcopy(self.context.data)
|
|
data["speaker_mappings"] = {
|
|
"Martin": data["participants"][0]["participant_id"]
|
|
}
|
|
with self.assertRaisesRegex(MeetingContextValidationError, "speaker label"):
|
|
validate_meeting_context(data)
|
|
|
|
def test_generated_context_yaml_is_deterministic_and_round_trips(self) -> None:
|
|
first = serialize_meeting_context_yaml(self.context)
|
|
second = serialize_meeting_context_yaml(self.context)
|
|
self.assertEqual(first, second)
|
|
|
|
SCRATCH_DIR.mkdir(exist_ok=True)
|
|
path = SCRATCH_DIR / "generated_context.yaml"
|
|
path.write_text(first, encoding="utf-8")
|
|
loaded = load_meeting_context(path)
|
|
self.assertEqual(loaded.data, self.context.data)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|