Add evidence-near semantic architecture experiments
Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
This commit is contained in:
@@ -0,0 +1,292 @@
|
||||
import json
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
from src.meeting_lab.topic_reconstruction.experiment import (
|
||||
ReconstructionValidationError,
|
||||
SCHEMA_VERSION,
|
||||
build_ollama_payload,
|
||||
evaluate_reconstruction,
|
||||
run_case,
|
||||
validate_evidence_units,
|
||||
validate_reconstruction,
|
||||
)
|
||||
|
||||
|
||||
class TopicReconstructionExperimentTests(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self.evidence = [
|
||||
{"evidence_id": "e1", "text": "Eine Variante wird vorgeschlagen."},
|
||||
{"evidence_id": "e2", "text": "Die Variante wird nur getestet."},
|
||||
{"evidence_id": "e3", "text": "Nina übernimmt die Prüfung."},
|
||||
{"evidence_id": "e4", "text": "Die Freigabe bleibt ungeklärt."},
|
||||
]
|
||||
|
||||
def valid_output(self):
|
||||
return {
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"subjects": [
|
||||
{
|
||||
"subject_id": "subject_1",
|
||||
"title": "Versuch mit der Variante",
|
||||
"evidence_refs": ["e1", "e2", "e3", "e4"],
|
||||
"development": [
|
||||
{
|
||||
"event_id": "event_1",
|
||||
"type": "proposal",
|
||||
"text": "Die Variante wurde für einen Versuch vorgeschlagen.",
|
||||
"evidence_refs": ["e1"],
|
||||
}
|
||||
],
|
||||
"outcome": {
|
||||
"text": "Die Variante wird getestet.",
|
||||
"scope": "Nur für den Versuch, nicht als endgültige Lösung.",
|
||||
"certainty": "established",
|
||||
"evidence_refs": ["e2"],
|
||||
},
|
||||
"actions": [
|
||||
{
|
||||
"action_id": "action_1",
|
||||
"text": "Die Variante prüfen.",
|
||||
"responsible": "Nina",
|
||||
"deadline": None,
|
||||
"evidence_refs": ["e3"],
|
||||
}
|
||||
],
|
||||
"unresolved_issues": [
|
||||
{
|
||||
"issue_id": "issue_1",
|
||||
"text": "Die Freigabe ist ungeklärt.",
|
||||
"evidence_refs": ["e4"],
|
||||
}
|
||||
],
|
||||
}
|
||||
],
|
||||
}
|
||||
|
||||
def test_schema_validation_accepts_sparse_subject(self):
|
||||
output = {
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"subjects": [
|
||||
{
|
||||
"subject_id": "subject_1",
|
||||
"title": "Geometrie",
|
||||
"evidence_refs": ["e1"],
|
||||
}
|
||||
],
|
||||
}
|
||||
|
||||
self.assertIs(validate_reconstruction(output, self.evidence), output)
|
||||
|
||||
def test_schema_validation_accepts_complete_structures(self):
|
||||
output = self.valid_output()
|
||||
|
||||
self.assertIs(validate_reconstruction(output, self.evidence), output)
|
||||
|
||||
def test_every_semantic_structure_requires_evidence_traceability(self):
|
||||
structures = [
|
||||
("subject", lambda data: data["subjects"][0].update(evidence_refs=[])),
|
||||
(
|
||||
"event",
|
||||
lambda data: data["subjects"][0]["development"][0].update(
|
||||
evidence_refs=[]
|
||||
),
|
||||
),
|
||||
(
|
||||
"outcome",
|
||||
lambda data: data["subjects"][0]["outcome"].update(evidence_refs=[]),
|
||||
),
|
||||
(
|
||||
"action",
|
||||
lambda data: data["subjects"][0]["actions"][0].update(
|
||||
evidence_refs=[]
|
||||
),
|
||||
),
|
||||
(
|
||||
"unresolved",
|
||||
lambda data: data["subjects"][0]["unresolved_issues"][0].update(
|
||||
evidence_refs=[]
|
||||
),
|
||||
),
|
||||
]
|
||||
for name, mutate in structures:
|
||||
with self.subTest(name=name):
|
||||
data = self.valid_output()
|
||||
mutate(data)
|
||||
with self.assertRaisesRegex(
|
||||
ReconstructionValidationError, "non-empty list"
|
||||
):
|
||||
validate_reconstruction(data, self.evidence)
|
||||
|
||||
def test_unknown_evidence_reference_is_rejected(self):
|
||||
output = self.valid_output()
|
||||
output["subjects"][0]["outcome"]["evidence_refs"] = ["e999"]
|
||||
|
||||
with self.assertRaisesRegex(
|
||||
ReconstructionValidationError, "unknown evidence ID: e999"
|
||||
):
|
||||
validate_reconstruction(output, self.evidence)
|
||||
|
||||
def test_duplicate_semantic_identifier_is_rejected(self):
|
||||
output = self.valid_output()
|
||||
output["subjects"][0]["actions"][0]["action_id"] = "event_1"
|
||||
|
||||
with self.assertRaisesRegex(
|
||||
ReconstructionValidationError, "duplicate identifier: event_1"
|
||||
):
|
||||
validate_reconstruction(output, self.evidence)
|
||||
|
||||
def test_duplicate_input_evidence_identifier_is_rejected(self):
|
||||
evidence = self.evidence + [
|
||||
{"evidence_id": "e1", "text": "Duplicate source."}
|
||||
]
|
||||
|
||||
with self.assertRaisesRegex(
|
||||
ReconstructionValidationError, "duplicate input evidence identifier"
|
||||
):
|
||||
validate_evidence_units(evidence)
|
||||
|
||||
def test_empty_subjects_are_rejected(self):
|
||||
output = {"schema_version": SCHEMA_VERSION, "subjects": []}
|
||||
|
||||
with self.assertRaisesRegex(
|
||||
ReconstructionValidationError, "subjects must be a non-empty list"
|
||||
):
|
||||
validate_reconstruction(output, self.evidence)
|
||||
|
||||
def test_blank_subject_title_is_rejected(self):
|
||||
output = self.valid_output()
|
||||
output["subjects"][0]["title"] = " "
|
||||
|
||||
with self.assertRaisesRegex(
|
||||
ReconstructionValidationError, "title must be a non-empty string"
|
||||
):
|
||||
validate_reconstruction(output, self.evidence)
|
||||
|
||||
def test_empty_optional_structures_must_be_omitted(self):
|
||||
for field, value in (
|
||||
("development", []),
|
||||
("outcome", None),
|
||||
("actions", []),
|
||||
("unresolved_issues", []),
|
||||
):
|
||||
with self.subTest(field=field):
|
||||
output = {
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"subjects": [
|
||||
{
|
||||
"subject_id": "subject_1",
|
||||
"title": "Subject",
|
||||
"evidence_refs": ["e1"],
|
||||
field: value,
|
||||
}
|
||||
],
|
||||
}
|
||||
with self.assertRaises(ReconstructionValidationError):
|
||||
validate_reconstruction(output, self.evidence)
|
||||
|
||||
def test_outcome_requires_scope_and_valid_certainty(self):
|
||||
output = self.valid_output()
|
||||
output["subjects"][0]["outcome"]["scope"] = ""
|
||||
with self.assertRaisesRegex(ReconstructionValidationError, "scope"):
|
||||
validate_reconstruction(output, self.evidence)
|
||||
|
||||
output = self.valid_output()
|
||||
output["subjects"][0]["outcome"]["certainty"] = "accepted_forever"
|
||||
with self.assertRaisesRegex(ReconstructionValidationError, "certainty"):
|
||||
validate_reconstruction(output, self.evidence)
|
||||
|
||||
def test_action_nullable_fields_and_unresolved_structure_are_strict(self):
|
||||
output = self.valid_output()
|
||||
output["subjects"][0]["actions"][0]["responsible"] = None
|
||||
validate_reconstruction(output, self.evidence)
|
||||
|
||||
output["subjects"][0]["unresolved_issues"][0]["extra"] = "invented"
|
||||
with self.assertRaisesRegex(ReconstructionValidationError, "unknown keys"):
|
||||
validate_reconstruction(output, self.evidence)
|
||||
|
||||
def test_action_nullable_fields_reject_string_null(self):
|
||||
output = self.valid_output()
|
||||
output["subjects"][0]["actions"][0]["responsible"] = "null"
|
||||
|
||||
with self.assertRaisesRegex(ReconstructionValidationError, "JSON null"):
|
||||
validate_reconstruction(output, self.evidence)
|
||||
|
||||
def test_ollama_payload_is_bounded_and_disables_thinking(self):
|
||||
payload = build_ollama_payload("qwen3.5:9B", "prompt", 16384, 4096)
|
||||
|
||||
self.assertEqual(payload["model"], "qwen3.5:9B")
|
||||
self.assertEqual(payload["format"], "json")
|
||||
self.assertIs(payload["stream"], False)
|
||||
self.assertIs(payload["think"], False)
|
||||
self.assertEqual(payload["options"]["temperature"], 0)
|
||||
self.assertEqual(payload["options"]["num_ctx"], 16384)
|
||||
self.assertEqual(payload["options"]["num_predict"], 4096)
|
||||
|
||||
def test_evaluator_marks_invented_action_as_critical_failure(self):
|
||||
output = self.valid_output()
|
||||
expected = {
|
||||
"subject_count": 1,
|
||||
"subject_terms": ["variante"],
|
||||
"required_event_types": ["proposal"],
|
||||
"outcome": {
|
||||
"required": True,
|
||||
"terms": ["getestet"],
|
||||
"scope_terms": ["nur"],
|
||||
"certainties": ["established"],
|
||||
},
|
||||
"actions": {"minimum": 0},
|
||||
"unresolved": {"minimum": 1, "terms": ["freigabe"]},
|
||||
}
|
||||
|
||||
result = evaluate_reconstruction(output, expected)
|
||||
|
||||
self.assertEqual(result["verdict"], "FAIL")
|
||||
self.assertIn("action_count", result["critical_failures"])
|
||||
|
||||
def test_validation_failure_preserves_inspection_artifacts(self):
|
||||
invalid = self.valid_output()
|
||||
invalid["subjects"][0]["outcome"]["evidence_refs"] = ["unknown"]
|
||||
raw = json.dumps(invalid, ensure_ascii=False)
|
||||
case = {
|
||||
"case_id": "artifact_case",
|
||||
"description": "Artifact preservation test.",
|
||||
"evidence_units": self.evidence,
|
||||
"expected": {},
|
||||
}
|
||||
metadata = {"elapsed_seconds": 0.01}
|
||||
|
||||
with tempfile.TemporaryDirectory() as temporary:
|
||||
root = Path(temporary)
|
||||
with patch(
|
||||
"src.meeting_lab.topic_reconstruction.experiment.call_ollama",
|
||||
return_value=(raw, metadata),
|
||||
):
|
||||
result = run_case(
|
||||
case,
|
||||
root,
|
||||
"http://unused",
|
||||
"qwen3.5:9B",
|
||||
1,
|
||||
1024,
|
||||
256,
|
||||
)
|
||||
|
||||
case_dir = root / "artifact_case"
|
||||
self.assertEqual(result["verdict"], "FAIL")
|
||||
self.assertIn("schema_validation", result["critical_failures"])
|
||||
self.assertTrue((case_dir / "input.json").exists())
|
||||
self.assertTrue((case_dir / "prompt.txt").exists())
|
||||
self.assertTrue((case_dir / "raw_model_response.txt").exists())
|
||||
self.assertTrue((case_dir / "parsed_output.json").exists())
|
||||
self.assertTrue((case_dir / "ollama_metadata.json").exists())
|
||||
failure = json.loads(
|
||||
(case_dir / "validation_failure.json").read_text(encoding="utf-8")
|
||||
)
|
||||
self.assertEqual(failure["error_type"], "ReconstructionValidationError")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user