Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
251 lines
9.5 KiB
Python
251 lines
9.5 KiB
Python
import json
|
|
import tempfile
|
|
import unittest
|
|
from pathlib import Path
|
|
from unittest.mock import patch
|
|
|
|
from src.meeting_lab.semantic_synthesis.experiment import (
|
|
SCHEMA_VERSION,
|
|
SynthesisValidationError,
|
|
build_ollama_payload,
|
|
evaluate_synthesis,
|
|
load_fixture,
|
|
run_case,
|
|
validate_bundle,
|
|
validate_synthesis,
|
|
)
|
|
|
|
|
|
class SemanticSynthesisExperimentTests(unittest.TestCase):
|
|
def setUp(self) -> None:
|
|
self.case = {
|
|
"case_id": "case_1",
|
|
"description": "Known subject test.",
|
|
"subject_id": "subject_1",
|
|
"subject": "Prüfung der Messdaten",
|
|
"evidence": [
|
|
{"evidence_id": "e1", "text": "Nina übernimmt die Prüfung."},
|
|
{"evidence_id": "e2", "text": "Die Freigabe bleibt offen."},
|
|
],
|
|
"allowed_responsible": ["Nina"],
|
|
"expected": {
|
|
"event_type_minimums": {"proposal": 1},
|
|
"allowed_event_types": ["proposal"],
|
|
"event_evidence_ids": ["e1"],
|
|
"outcome": {
|
|
"required": True,
|
|
"statuses": ["established"],
|
|
"terms": ["prüfung"],
|
|
"scope_terms": ["messdaten"],
|
|
"evidence_ids": ["e1"],
|
|
},
|
|
"actions": {
|
|
"count": 1,
|
|
"terms": ["prüfung"],
|
|
"responsible": "Nina",
|
|
"due_terms": [],
|
|
"evidence_ids": ["e1"],
|
|
},
|
|
"unresolved_issues": {
|
|
"count": 1,
|
|
"terms": ["freigabe"],
|
|
"evidence_ids": ["e2"],
|
|
},
|
|
},
|
|
}
|
|
|
|
def valid_output(self):
|
|
return {
|
|
"schema_version": SCHEMA_VERSION,
|
|
"subject_id": "subject_1",
|
|
"subject": "Prüfung der Messdaten",
|
|
"events": [
|
|
{
|
|
"type": "proposal",
|
|
"text": "Die Prüfung wird vorgeschlagen.",
|
|
"evidence_ids": ["e1"],
|
|
}
|
|
],
|
|
"outcome": {
|
|
"status": "established",
|
|
"text": "Die Prüfung wird übernommen.",
|
|
"scope": "Prüfung der Messdaten",
|
|
"evidence_ids": ["e1"],
|
|
},
|
|
"actions": [
|
|
{
|
|
"text": "Prüfung der Messdaten durchführen.",
|
|
"responsible": "Nina",
|
|
"due": None,
|
|
"evidence_ids": ["e1"],
|
|
}
|
|
],
|
|
"unresolved_issues": [
|
|
{
|
|
"text": "Die Freigabe bleibt offen.",
|
|
"evidence_ids": ["e2"],
|
|
}
|
|
],
|
|
}
|
|
|
|
def test_bundle_validation_accepts_fixed_subject_and_complete_evidence(self):
|
|
self.assertIs(validate_bundle(self.case), self.case)
|
|
|
|
def test_bundle_validation_rejects_duplicate_evidence_ids(self):
|
|
case = dict(self.case)
|
|
case["evidence"] = self.case["evidence"] * 2
|
|
with self.assertRaisesRegex(SynthesisValidationError, "duplicate evidence ID"):
|
|
validate_bundle(case)
|
|
|
|
def test_sparse_absence_uses_empty_arrays_and_omitted_outcome(self):
|
|
output = {
|
|
"schema_version": SCHEMA_VERSION,
|
|
"subject_id": "subject_1",
|
|
"subject": "Prüfung der Messdaten",
|
|
"events": [],
|
|
"actions": [],
|
|
"unresolved_issues": [],
|
|
}
|
|
|
|
self.assertIs(validate_synthesis(output, self.case), output)
|
|
|
|
def test_outcome_null_is_rejected_but_omission_is_allowed(self):
|
|
output = self.valid_output()
|
|
output["outcome"] = None
|
|
with self.assertRaisesRegex(SynthesisValidationError, "omit it when absent"):
|
|
validate_synthesis(output, self.case)
|
|
|
|
def test_required_arrays_must_exist(self):
|
|
for field in ("events", "actions", "unresolved_issues"):
|
|
with self.subTest(field=field):
|
|
output = self.valid_output()
|
|
del output[field]
|
|
with self.assertRaisesRegex(SynthesisValidationError, "missing required"):
|
|
validate_synthesis(output, self.case)
|
|
|
|
def test_fixed_subject_identity_cannot_change(self):
|
|
output = self.valid_output()
|
|
output["subject"] = "Different subject"
|
|
with self.assertRaisesRegex(SynthesisValidationError, "changed fixed subject"):
|
|
validate_synthesis(output, self.case)
|
|
|
|
def test_unknown_evidence_id_is_rejected_in_every_structure(self):
|
|
mutations = (
|
|
lambda output: output["events"][0].update(evidence_ids=["unknown"]),
|
|
lambda output: output["outcome"].update(evidence_ids=["unknown"]),
|
|
lambda output: output["actions"][0].update(evidence_ids=["unknown"]),
|
|
lambda output: output["unresolved_issues"][0].update(
|
|
evidence_ids=["unknown"]
|
|
),
|
|
)
|
|
for mutate in mutations:
|
|
output = self.valid_output()
|
|
mutate(output)
|
|
with self.assertRaisesRegex(SynthesisValidationError, "unknown evidence ID"):
|
|
validate_synthesis(output, self.case)
|
|
|
|
def test_duplicate_evidence_reference_is_rejected(self):
|
|
output = self.valid_output()
|
|
output["events"][0]["evidence_ids"] = ["e1", "e1"]
|
|
with self.assertRaisesRegex(SynthesisValidationError, "duplicate evidence ID"):
|
|
validate_synthesis(output, self.case)
|
|
|
|
def test_responsibility_must_be_allowed_or_json_null(self):
|
|
output = self.valid_output()
|
|
output["actions"][0]["responsible"] = None
|
|
validate_synthesis(output, self.case)
|
|
|
|
output["actions"][0]["responsible"] = "Martin"
|
|
with self.assertRaisesRegex(SynthesisValidationError, "not allowed"):
|
|
validate_synthesis(output, self.case)
|
|
|
|
def test_string_null_is_rejected(self):
|
|
output = self.valid_output()
|
|
output["actions"][0]["due"] = "null"
|
|
with self.assertRaisesRegex(SynthesisValidationError, "JSON null"):
|
|
validate_synthesis(output, self.case)
|
|
|
|
def test_outcome_action_and_unresolved_structures_are_strict(self):
|
|
for field, target in (
|
|
("extra", lambda output: output["outcome"]),
|
|
("extra", lambda output: output["actions"][0]),
|
|
("extra", lambda output: output["unresolved_issues"][0]),
|
|
):
|
|
output = self.valid_output()
|
|
target(output)[field] = "not allowed"
|
|
with self.assertRaisesRegex(SynthesisValidationError, "unknown keys"):
|
|
validate_synthesis(output, self.case)
|
|
|
|
def test_evaluator_passes_complete_semantics(self):
|
|
result = evaluate_synthesis(self.valid_output(), self.case["expected"])
|
|
self.assertEqual(result["verdict"], "PASS")
|
|
|
|
def test_evaluator_treats_invented_action_as_critical(self):
|
|
output = self.valid_output()
|
|
expected = dict(self.case["expected"])
|
|
expected["actions"] = {"count": 0}
|
|
result = evaluate_synthesis(output, expected)
|
|
self.assertEqual(result["verdict"], "FAIL")
|
|
self.assertIn("action_count", result["critical_failures"])
|
|
|
|
def test_ollama_payload_is_bounded_and_has_required_controls(self):
|
|
payload = build_ollama_payload("qwen3.5:9B", "prompt", 8192, 2048)
|
|
self.assertEqual(payload["format"], "json")
|
|
self.assertIs(payload["think"], False)
|
|
self.assertIs(payload["stream"], False)
|
|
self.assertEqual(payload["options"]["temperature"], 0)
|
|
self.assertEqual(payload["options"]["num_ctx"], 8192)
|
|
self.assertEqual(payload["options"]["num_predict"], 2048)
|
|
|
|
def test_fixture_contains_all_nine_isolation_cases(self):
|
|
cases = load_fixture(Path("tests/gold/semantic_synthesis_isolation/cases.json"))
|
|
self.assertEqual(
|
|
[case["case_id"] for case in cases],
|
|
[
|
|
"a_idea_only",
|
|
"b_multiple_options",
|
|
"c_unaccepted_proposal",
|
|
"d_proposal_with_objection",
|
|
"e_rejected_alternative",
|
|
"f_trial_only_acceptance",
|
|
"g_no_decision",
|
|
"h_resulting_action",
|
|
"i_outcome_and_unresolved",
|
|
],
|
|
)
|
|
|
|
def test_case_run_preserves_all_inspection_artifacts(self):
|
|
raw = json.dumps(self.valid_output(), ensure_ascii=False)
|
|
metadata = {"elapsed_seconds": 0.01}
|
|
with tempfile.TemporaryDirectory() as temporary:
|
|
root = Path(temporary)
|
|
with patch(
|
|
"src.meeting_lab.semantic_synthesis.experiment.call_ollama",
|
|
return_value=(raw, metadata),
|
|
):
|
|
result = run_case(
|
|
self.case,
|
|
root,
|
|
"http://unused",
|
|
"qwen3.5:9B",
|
|
1,
|
|
8192,
|
|
2048,
|
|
)
|
|
self.assertEqual(result["verdict"], "PASS")
|
|
case_dir = root / "case_1"
|
|
for filename in (
|
|
"gold_input.json",
|
|
"prompt.txt",
|
|
"raw_model_response.txt",
|
|
"parsed_response.json",
|
|
"ollama_metadata.json",
|
|
"validation_result.json",
|
|
"evaluation_result.json",
|
|
):
|
|
self.assertTrue((case_dir / filename).exists(), filename)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|