Stabilize Meeting Lab pipeline for RC1 evaluation
This commit significantly improves the robustness and determinism of the Meeting Lab processing pipeline and establishes the first Release Candidate baseline for end-to-end evaluation. Highlights - BUG-009 - Implement deterministic responsible-party validation - Normalize participant aliases using Meeting Context - Reject invalid responsible values (dates, locations, technical terms, projects, products, unknown entities) - Record structured responsibility validation metadata - Add focused regression tests - BUG-010 - Implement adaptive num_predict estimation for Semantic Consolidator - Eliminate JSON truncation caused by fixed output limits - Add deterministic source coverage repair - Preserve strict post-repair validation - Add regression tests - BUG-011 - Implement Working Protocol V2 renderer contract enforcement - Preserve raw renderer responses - Reject invalid protocol output instead of accepting malformed documents - Add deterministic cleanup for harmless formatting deviations - Add focused renderer regression tests - Meeting Context - Validate Meeting Context V1 - Integrate authoritative participant alias normalization - Documentation - Update architecture documentation - Update output documentation - Update regression bug tracker The pipeline now fails safely instead of silently accepting invalid intermediate or final artifacts. Remaining work focuses primarily on extraction quality and semantic classification (decisions, action items, protocol faithfulness), rather than pipeline robustness.
This commit is contained in:
@@ -0,0 +1,179 @@
|
||||
import json
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
from src.meeting_lab.protocol.render_working_protocol import (
|
||||
REQUIRED_TITLE,
|
||||
clean_working_protocol_markdown,
|
||||
render_working_protocol,
|
||||
validate_working_protocol_markdown,
|
||||
)
|
||||
|
||||
|
||||
VALID_PROTOCOL = """# Working Protocol
|
||||
|
||||
## Materialversuche
|
||||
|
||||
### Background
|
||||
|
||||
Die Materialversuche wurden besprochen.
|
||||
|
||||
### Decisions
|
||||
|
||||
- Der nächste Versuch wird vorbereitet.
|
||||
|
||||
### Action Items
|
||||
|
||||
- Verantwortung offen: Materialstatus prüfen.
|
||||
|
||||
### Open Questions
|
||||
|
||||
- Wann liegen die Ergebnisse vor?
|
||||
"""
|
||||
|
||||
|
||||
class WorkingProtocolRendererTests(unittest.TestCase):
|
||||
def test_valid_protocol_beginning_with_required_heading_passes(self) -> None:
|
||||
report = validate_working_protocol_markdown(VALID_PROTOCOL)
|
||||
|
||||
self.assertTrue(report.valid)
|
||||
self.assertEqual(report.violations, [])
|
||||
|
||||
def test_explanatory_prose_before_valid_protocol_is_removed(self) -> None:
|
||||
cleaned = clean_working_protocol_markdown(
|
||||
"Hier ist das Protokoll:\n\n" + VALID_PROTOCOL
|
||||
)
|
||||
|
||||
self.assertTrue(cleaned.startswith(REQUIRED_TITLE + "\n"))
|
||||
self.assertNotIn("Hier ist das Protokoll", cleaned)
|
||||
self.assertTrue(validate_working_protocol_markdown(cleaned).valid)
|
||||
|
||||
def test_leading_whitespace_before_valid_heading_is_removed(self) -> None:
|
||||
cleaned = clean_working_protocol_markdown("\n \n # Working Protocol\n\n## Thema\n\n### Background\n\nText.\n")
|
||||
|
||||
self.assertTrue(cleaned.startswith(REQUIRED_TITLE + "\n"))
|
||||
self.assertTrue(validate_working_protocol_markdown(cleaned).valid)
|
||||
|
||||
def test_output_without_required_heading_is_rejected(self) -> None:
|
||||
report = validate_working_protocol_markdown("## Materialversuche\n\nText.\n")
|
||||
|
||||
self.assertFalse(report.valid)
|
||||
self.assertEqual(report.violations[0]["type"], "missing_required_heading")
|
||||
|
||||
def test_categorized_summary_without_topic_body_is_rejected(self) -> None:
|
||||
text = """Hier ist die Zusammenfassung:
|
||||
|
||||
### **Entscheidungen (Decisions)**
|
||||
|
||||
- Entscheidung.
|
||||
|
||||
### **Handlungsaufträge (Action Items)**
|
||||
|
||||
- Aufgabe.
|
||||
"""
|
||||
|
||||
cleaned = clean_working_protocol_markdown(text)
|
||||
report = validate_working_protocol_markdown(cleaned)
|
||||
|
||||
self.assertFalse(report.valid)
|
||||
self.assertEqual(report.violations[0]["type"], "missing_required_heading")
|
||||
|
||||
def test_categorized_summary_after_heading_is_rejected(self) -> None:
|
||||
text = """# Working Protocol
|
||||
|
||||
## Entscheidungen
|
||||
|
||||
- Entscheidung.
|
||||
|
||||
## Action Items
|
||||
|
||||
- Aufgabe.
|
||||
"""
|
||||
|
||||
report = validate_working_protocol_markdown(text)
|
||||
|
||||
self.assertFalse(report.valid)
|
||||
violation_types = {violation["type"] for violation in report.violations}
|
||||
self.assertIn("generic_summary_framing", violation_types)
|
||||
self.assertIn("missing_topic_sections", violation_types)
|
||||
|
||||
def test_malformed_markdown_is_rejected(self) -> None:
|
||||
text = """# Working Protocol
|
||||
|
||||
## Materialversuche
|
||||
|
||||
### Background
|
||||
|
||||
```json
|
||||
{"unterbrochen": true}
|
||||
"""
|
||||
|
||||
report = validate_working_protocol_markdown(text)
|
||||
|
||||
self.assertFalse(report.valid)
|
||||
self.assertIn(
|
||||
"malformed_markdown",
|
||||
{violation["type"] for violation in report.violations},
|
||||
)
|
||||
|
||||
def test_raw_response_preserved_and_protocol_contains_only_validated_content(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
input_path = root / "input.json"
|
||||
input_path.write_text('{"items": []}\n', encoding="utf-8")
|
||||
output_dir = root / "working_protocol"
|
||||
|
||||
with patch(
|
||||
"src.meeting_lab.protocol.render_working_protocol.call_ollama",
|
||||
return_value=(
|
||||
{
|
||||
"response": "Einleitung.\n\n" + VALID_PROTOCOL,
|
||||
"done": True,
|
||||
"done_reason": "stop",
|
||||
},
|
||||
1.25,
|
||||
),
|
||||
):
|
||||
metadata = render_working_protocol(input_path, output_dir)
|
||||
|
||||
raw = json.loads((output_dir / "raw_model_response.json").read_text(encoding="utf-8"))
|
||||
self.assertEqual(raw["response"], "Einleitung.\n\n" + VALID_PROTOCOL)
|
||||
self.assertTrue((output_dir / "cleaned_candidate.md").exists())
|
||||
self.assertTrue((output_dir / "validation_report.json").exists())
|
||||
self.assertTrue(metadata["valid"])
|
||||
|
||||
protocol = (output_dir / "working_protocol.md").read_text(encoding="utf-8")
|
||||
self.assertEqual(protocol, clean_working_protocol_markdown(raw["response"]))
|
||||
self.assertNotIn("Einleitung.", protocol)
|
||||
|
||||
def test_invalid_output_preserves_raw_but_does_not_write_protocol(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
input_path = root / "input.json"
|
||||
input_path.write_text('{"items": []}\n', encoding="utf-8")
|
||||
output_dir = root / "working_protocol"
|
||||
|
||||
with patch(
|
||||
"src.meeting_lab.protocol.render_working_protocol.call_ollama",
|
||||
return_value=(
|
||||
{
|
||||
"response": "Hier ist die Zusammenfassung:\n\n### Entscheidungen\n\n- X\n",
|
||||
"done": True,
|
||||
"done_reason": "stop",
|
||||
},
|
||||
1.25,
|
||||
),
|
||||
):
|
||||
metadata = render_working_protocol(input_path, output_dir)
|
||||
|
||||
self.assertFalse(metadata["valid"])
|
||||
self.assertTrue((output_dir / "raw_model_response.json").exists())
|
||||
self.assertTrue((output_dir / "cleaned_candidate.md").exists())
|
||||
self.assertTrue((output_dir / "validation_report.json").exists())
|
||||
self.assertFalse((output_dir / "working_protocol.md").exists())
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user