- introduce Gold Standard evaluation corpus - document decision taxonomy - define prompt-engineering methodology - add regression workflow - establish Prompt Version 2 baseline - validate decision_simple, decision_deferred and decision_none
134 lines
4.4 KiB
Python
134 lines
4.4 KiB
Python
import json
|
|
import tempfile
|
|
import unittest
|
|
from pathlib import Path
|
|
|
|
from src.meeting_lab.extraction.extract_chunks import (
|
|
EXTRACTION_CATEGORIES,
|
|
build_prompt,
|
|
extraction_path_for_chunk,
|
|
normalize_current_schema,
|
|
parse_json_response,
|
|
)
|
|
from src.meeting_lab.protocol.build_protocol import build_protocol
|
|
|
|
|
|
class ExtractionProtocolTests(unittest.TestCase):
|
|
def test_extraction_path_drops_normalized_suffix(self) -> None:
|
|
path = Path("chunks/chunk_01_normalized.txt")
|
|
|
|
self.assertEqual(
|
|
extraction_path_for_chunk(path),
|
|
Path("chunks/chunk_01_extraction.json"),
|
|
)
|
|
|
|
def test_normalize_current_schema_maps_legacy_response(self) -> None:
|
|
result = normalize_current_schema(
|
|
{
|
|
"facts": [
|
|
{
|
|
"speaker": "A",
|
|
"statement": "Fact one",
|
|
"status": "clear",
|
|
"evidence": "Fact",
|
|
}
|
|
],
|
|
"decisions": [
|
|
{
|
|
"decision": "Decision one",
|
|
"evidence": "Decision",
|
|
}
|
|
],
|
|
"todos": [
|
|
{
|
|
"task": "Todo one",
|
|
"responsible": "B",
|
|
"deadline": None,
|
|
"evidence": "Todo",
|
|
}
|
|
],
|
|
"open_questions": [
|
|
{
|
|
"question": "Question one",
|
|
"evidence": "Question",
|
|
}
|
|
],
|
|
"technical_details": [
|
|
{
|
|
"subject": "System",
|
|
"statement": "Technical one",
|
|
"status": "clear",
|
|
"evidence": "Technical",
|
|
}
|
|
],
|
|
}
|
|
)
|
|
|
|
self.assertEqual(set(result), set(EXTRACTION_CATEGORIES))
|
|
self.assertEqual(result["positions"], [])
|
|
self.assertIn("Fact one", result["facts"][0])
|
|
self.assertIn("Decision one", result["decisions"][0])
|
|
self.assertIn("Todo one", result["todos"][0])
|
|
self.assertIn("Question one", result["questions"][0])
|
|
self.assertIn("Technical one", result["technical"][0])
|
|
|
|
def test_parse_json_response_uses_final_object_after_thinking(self) -> None:
|
|
text = """
|
|
Thinking:
|
|
I will reason about the transcript first.
|
|
{"facts": ["draft"], "decisions": []}
|
|
|
|
Final answer:
|
|
{
|
|
"facts": ["final"],
|
|
"decisions": [],
|
|
"todos": [],
|
|
"questions": [],
|
|
"positions": [],
|
|
"technical": []
|
|
}
|
|
"""
|
|
|
|
parsed = parse_json_response(text)
|
|
|
|
self.assertEqual(parsed["facts"], ["final"])
|
|
self.assertEqual(set(parsed), set(EXTRACTION_CATEGORIES))
|
|
|
|
def test_build_prompt_includes_common_and_decision_prompt_files(self) -> None:
|
|
prompt = build_prompt("transcript.txt", "Anna: Agreed.")
|
|
|
|
self.assertIn("Du extrahierst Informationen aus Meeting-Transkripten.", prompt)
|
|
self.assertIn("You extract decisions from meeting transcript text.", prompt)
|
|
self.assertIn("Extract each decision as one atomic commitment.", prompt)
|
|
self.assertIn("Anna: Agreed.", prompt)
|
|
|
|
def test_build_protocol_groups_extraction_items(self) -> None:
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
input_dir = Path(directory)
|
|
(input_dir / "chunk_01_extraction.json").write_text(
|
|
json.dumps(
|
|
{
|
|
"facts": ["Fact one"],
|
|
"decisions": ["Decision one"],
|
|
"todos": ["Todo one"],
|
|
"questions": ["Question one"],
|
|
"positions": [],
|
|
"technical": [],
|
|
}
|
|
),
|
|
encoding="utf-8",
|
|
)
|
|
|
|
output_path = build_protocol(input_dir)
|
|
|
|
markdown = output_path.read_text(encoding="utf-8")
|
|
self.assertIn("# Meeting Protocol", markdown)
|
|
self.assertIn("## Facts\n- Fact one", markdown)
|
|
self.assertIn("## Decisions\n- Decision one", markdown)
|
|
self.assertIn("## Open Questions\n- Question one", markdown)
|
|
self.assertIn("## Action Items\n- Todo one", markdown)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|