Add evidence-near semantic architecture experiments
Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
This commit is contained in:
@@ -0,0 +1,114 @@
|
||||
import json
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
from src.meeting_lab.evidence_observations_v3.experiment import (
|
||||
SCHEMA_VERSION,
|
||||
ObservationValidationError,
|
||||
build_ollama_payload,
|
||||
load_fixture,
|
||||
parse_model_json,
|
||||
run_case,
|
||||
validate_observations,
|
||||
)
|
||||
|
||||
|
||||
class EvidenceObservationV3ExperimentTests(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self.case = {
|
||||
"case_id": "test", "description": "Minimal fixture.",
|
||||
"subject_id": "subject_test", "subject": "Messdatenprüfung",
|
||||
"evidence": [
|
||||
{"evidence_id": "e1", "text": "Antonius: Nina, prüfst du die Messdaten?"},
|
||||
{"evidence_id": "e2", "text": "Nina: Ja, ich prüfe sie bis Freitag."},
|
||||
],
|
||||
"semantic_requirements": ["Request and response survive."],
|
||||
}
|
||||
self.output = {
|
||||
"schema_version": SCHEMA_VERSION, "subject_id": "subject_test",
|
||||
"subject": "Messdatenprüfung", "observations": [self.observation()],
|
||||
}
|
||||
|
||||
def observation(self, **updates):
|
||||
value = {
|
||||
"observation_id": "obs_1", "evidence_id": "e1",
|
||||
"content": "Antonius fragt Nina, ob sie die Messdaten prüft.",
|
||||
"speaker": "Antonius", "named_person": "Nina", "addressee": "Nina",
|
||||
}
|
||||
value.update(updates)
|
||||
return value
|
||||
|
||||
def test_minimal_schema_is_valid(self):
|
||||
self.assertIs(validate_observations(self.output, self.case), self.output)
|
||||
self.assertEqual(set(self.output["observations"][0]), {
|
||||
"observation_id", "evidence_id", "content", "speaker", "named_person", "addressee"
|
||||
})
|
||||
|
||||
def test_unknown_semantic_field_is_rejected(self):
|
||||
self.output["observations"][0]["modality"] = "factual"
|
||||
with self.assertRaisesRegex(ObservationValidationError, "unknown keys"):
|
||||
validate_observations(self.output, self.case)
|
||||
|
||||
def test_unknown_evidence_is_rejected(self):
|
||||
self.output["observations"][0]["evidence_id"] = "e9"
|
||||
with self.assertRaisesRegex(ObservationValidationError, "unknown evidence"):
|
||||
validate_observations(self.output, self.case)
|
||||
|
||||
def test_speaker_must_match_evidence(self):
|
||||
self.output["observations"][0]["speaker"] = "Nina"
|
||||
with self.assertRaisesRegex(ObservationValidationError, "match evidence speaker"):
|
||||
validate_observations(self.output, self.case)
|
||||
|
||||
def test_named_person_does_not_add_responsibility(self):
|
||||
validate_observations(self.output, self.case)
|
||||
self.assertNotIn("responsibility", self.output["observations"][0])
|
||||
|
||||
def test_addressee_does_not_add_assignment(self):
|
||||
validate_observations(self.output, self.case)
|
||||
self.assertNotIn("action_item", self.output["observations"][0])
|
||||
|
||||
def test_nonexplicit_person_is_rejected(self):
|
||||
self.output["observations"][0]["named_person"] = "Martin"
|
||||
with self.assertRaisesRegex(ObservationValidationError, "not an explicit person"):
|
||||
validate_observations(self.output, self.case)
|
||||
|
||||
def test_multiple_atomic_observations_can_share_evidence(self):
|
||||
self.output["observations"].append(self.observation(observation_id="obs_2"))
|
||||
validate_observations(self.output, self.case)
|
||||
|
||||
def test_malformed_json_is_rejected(self):
|
||||
with self.assertRaises(json.JSONDecodeError):
|
||||
parse_model_json("{bad json")
|
||||
|
||||
def test_payload_controls_are_fixed(self):
|
||||
payload = build_ollama_payload("qwen3.5:9B", "prompt", 16384, 4096)
|
||||
self.assertFalse(payload["think"])
|
||||
self.assertEqual(payload["options"]["temperature"], 0)
|
||||
|
||||
def test_fixture_reuses_exact_v2_evidence(self):
|
||||
v3 = load_fixture(Path("tests/gold/evidence_observations_v3/cases.json"))
|
||||
v2 = json.loads(Path("tests/gold/evidence_observations_v2/cases.json").read_text())["cases"]
|
||||
self.assertEqual([case["evidence"] for case in v3], [case["evidence"] for case in v2])
|
||||
|
||||
def test_case_run_preserves_persistent_artifact_set(self):
|
||||
raw = json.dumps(self.output, ensure_ascii=False)
|
||||
with tempfile.TemporaryDirectory() as temporary:
|
||||
root = Path(temporary)
|
||||
with patch(
|
||||
"src.meeting_lab.evidence_observations_v3.experiment.call_ollama",
|
||||
return_value=(raw, {"model": "qwen3.5:9B"}),
|
||||
):
|
||||
result = run_case(self.case, root, "http://unused", "qwen3.5:9B", 1, 16384, 4096)
|
||||
self.assertTrue(result["structurally_valid"])
|
||||
for filename in (
|
||||
"source_evidence.json", "gold_semantic_requirements.json", "prompt.txt",
|
||||
"raw_model_response.txt", "parsed_observations.json",
|
||||
"structural_validation.json", "ollama_metadata.json",
|
||||
):
|
||||
self.assertTrue((root / "test" / filename).is_file(), filename)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user