Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
115 lines
4.9 KiB
Python
115 lines
4.9 KiB
Python
import json
|
|
import tempfile
|
|
import unittest
|
|
from pathlib import Path
|
|
from unittest.mock import patch
|
|
|
|
from src.meeting_lab.evidence_observations_v3.experiment import (
|
|
SCHEMA_VERSION,
|
|
ObservationValidationError,
|
|
build_ollama_payload,
|
|
load_fixture,
|
|
parse_model_json,
|
|
run_case,
|
|
validate_observations,
|
|
)
|
|
|
|
|
|
class EvidenceObservationV3ExperimentTests(unittest.TestCase):
|
|
def setUp(self) -> None:
|
|
self.case = {
|
|
"case_id": "test", "description": "Minimal fixture.",
|
|
"subject_id": "subject_test", "subject": "Messdatenprüfung",
|
|
"evidence": [
|
|
{"evidence_id": "e1", "text": "Antonius: Nina, prüfst du die Messdaten?"},
|
|
{"evidence_id": "e2", "text": "Nina: Ja, ich prüfe sie bis Freitag."},
|
|
],
|
|
"semantic_requirements": ["Request and response survive."],
|
|
}
|
|
self.output = {
|
|
"schema_version": SCHEMA_VERSION, "subject_id": "subject_test",
|
|
"subject": "Messdatenprüfung", "observations": [self.observation()],
|
|
}
|
|
|
|
def observation(self, **updates):
|
|
value = {
|
|
"observation_id": "obs_1", "evidence_id": "e1",
|
|
"content": "Antonius fragt Nina, ob sie die Messdaten prüft.",
|
|
"speaker": "Antonius", "named_person": "Nina", "addressee": "Nina",
|
|
}
|
|
value.update(updates)
|
|
return value
|
|
|
|
def test_minimal_schema_is_valid(self):
|
|
self.assertIs(validate_observations(self.output, self.case), self.output)
|
|
self.assertEqual(set(self.output["observations"][0]), {
|
|
"observation_id", "evidence_id", "content", "speaker", "named_person", "addressee"
|
|
})
|
|
|
|
def test_unknown_semantic_field_is_rejected(self):
|
|
self.output["observations"][0]["modality"] = "factual"
|
|
with self.assertRaisesRegex(ObservationValidationError, "unknown keys"):
|
|
validate_observations(self.output, self.case)
|
|
|
|
def test_unknown_evidence_is_rejected(self):
|
|
self.output["observations"][0]["evidence_id"] = "e9"
|
|
with self.assertRaisesRegex(ObservationValidationError, "unknown evidence"):
|
|
validate_observations(self.output, self.case)
|
|
|
|
def test_speaker_must_match_evidence(self):
|
|
self.output["observations"][0]["speaker"] = "Nina"
|
|
with self.assertRaisesRegex(ObservationValidationError, "match evidence speaker"):
|
|
validate_observations(self.output, self.case)
|
|
|
|
def test_named_person_does_not_add_responsibility(self):
|
|
validate_observations(self.output, self.case)
|
|
self.assertNotIn("responsibility", self.output["observations"][0])
|
|
|
|
def test_addressee_does_not_add_assignment(self):
|
|
validate_observations(self.output, self.case)
|
|
self.assertNotIn("action_item", self.output["observations"][0])
|
|
|
|
def test_nonexplicit_person_is_rejected(self):
|
|
self.output["observations"][0]["named_person"] = "Martin"
|
|
with self.assertRaisesRegex(ObservationValidationError, "not an explicit person"):
|
|
validate_observations(self.output, self.case)
|
|
|
|
def test_multiple_atomic_observations_can_share_evidence(self):
|
|
self.output["observations"].append(self.observation(observation_id="obs_2"))
|
|
validate_observations(self.output, self.case)
|
|
|
|
def test_malformed_json_is_rejected(self):
|
|
with self.assertRaises(json.JSONDecodeError):
|
|
parse_model_json("{bad json")
|
|
|
|
def test_payload_controls_are_fixed(self):
|
|
payload = build_ollama_payload("qwen3.5:9B", "prompt", 16384, 4096)
|
|
self.assertFalse(payload["think"])
|
|
self.assertEqual(payload["options"]["temperature"], 0)
|
|
|
|
def test_fixture_reuses_exact_v2_evidence(self):
|
|
v3 = load_fixture(Path("tests/gold/evidence_observations_v3/cases.json"))
|
|
v2 = json.loads(Path("tests/gold/evidence_observations_v2/cases.json").read_text())["cases"]
|
|
self.assertEqual([case["evidence"] for case in v3], [case["evidence"] for case in v2])
|
|
|
|
def test_case_run_preserves_persistent_artifact_set(self):
|
|
raw = json.dumps(self.output, ensure_ascii=False)
|
|
with tempfile.TemporaryDirectory() as temporary:
|
|
root = Path(temporary)
|
|
with patch(
|
|
"src.meeting_lab.evidence_observations_v3.experiment.call_ollama",
|
|
return_value=(raw, {"model": "qwen3.5:9B"}),
|
|
):
|
|
result = run_case(self.case, root, "http://unused", "qwen3.5:9B", 1, 16384, 4096)
|
|
self.assertTrue(result["structurally_valid"])
|
|
for filename in (
|
|
"source_evidence.json", "gold_semantic_requirements.json", "prompt.txt",
|
|
"raw_model_response.txt", "parsed_observations.json",
|
|
"structural_validation.json", "ollama_metadata.json",
|
|
):
|
|
self.assertTrue((root / "test" / filename).is_file(), filename)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|