Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
161 lines
7.2 KiB
Python
161 lines
7.2 KiB
Python
import json
|
|
import tempfile
|
|
import unittest
|
|
from copy import deepcopy
|
|
from pathlib import Path
|
|
from unittest.mock import patch
|
|
|
|
from src.meeting_lab.evidence_observations_v2.experiment import (
|
|
SCHEMA_VERSION,
|
|
ObservationValidationError,
|
|
build_ollama_payload,
|
|
load_fixture,
|
|
parse_model_json,
|
|
run_case,
|
|
validate_observations,
|
|
)
|
|
|
|
|
|
class EvidenceObservationV2ExperimentTests(unittest.TestCase):
|
|
def setUp(self) -> None:
|
|
self.case = {
|
|
"case_id": "test_case", "description": "Validator fixture.",
|
|
"subject_id": "subject_test", "subject": "Prüfung der Messdaten",
|
|
"evidence": [
|
|
{"evidence_id": "e1", "text": "Antonius: Nina, prüfst du die Daten?"},
|
|
{"evidence_id": "e2", "text": "Nina: Ja, ich prüfe sie."},
|
|
],
|
|
"expected_observations": [],
|
|
}
|
|
self.output = {
|
|
"schema_version": SCHEMA_VERSION,
|
|
"subject_id": self.case["subject_id"], "subject": self.case["subject"],
|
|
"observations": [self.observation()],
|
|
}
|
|
self.case["expected_observations"] = deepcopy(self.output["observations"])
|
|
|
|
def observation(self, **updates):
|
|
value = {
|
|
"observation_id": "obs_1", "evidence_id": "e1",
|
|
"content": "Antonius bittet Nina um eine Prüfung.", "refers_to": None,
|
|
"speaker": "Antonius", "named_person": "Nina", "addressee": "Nina",
|
|
"self_reference": False, "collective_we": False,
|
|
"impersonal_person_reference": False,
|
|
"modality": "interpersonal_request", "temporality": "future",
|
|
"evaluation": "none", "affirmation": "absent", "negation": "absent",
|
|
"determination_statement": "absent", "uncertainty": "absent",
|
|
"clarification_need": "none", "qualifier": "bis Freitag",
|
|
"limits_target": None,
|
|
}
|
|
value.update(updates)
|
|
return value
|
|
|
|
def test_participant_facts_do_not_include_responsibility(self):
|
|
validate_observations(self.output, self.case)
|
|
observation = self.output["observations"][0]
|
|
self.assertEqual(observation["speaker"], "Antonius")
|
|
self.assertEqual(observation["named_person"], "Nina")
|
|
self.assertEqual(observation["addressee"], "Nina")
|
|
self.assertNotIn("responsibility", observation)
|
|
|
|
def test_named_person_and_speaker_do_not_imply_any_extra_field(self):
|
|
keys = self.output["observations"][0].keys()
|
|
self.assertNotIn("person", keys)
|
|
self.assertNotIn("agreement", keys)
|
|
|
|
def test_self_reference_collective_we_and_impersonal_reference_are_boolean(self):
|
|
self.output["observations"][0].update(
|
|
self_reference=True, collective_we=True, impersonal_person_reference=True
|
|
)
|
|
validate_observations(self.output, self.case)
|
|
self.output["observations"][0]["collective_we"] = "true"
|
|
with self.assertRaisesRegex(ObservationValidationError, "must be boolean"):
|
|
validate_observations(self.output, self.case)
|
|
|
|
def test_explicit_affirmation_negation_and_determination(self):
|
|
self.output["observations"][0].update(
|
|
affirmation="explicit", negation="explicit", determination_statement="present"
|
|
)
|
|
validate_observations(self.output, self.case)
|
|
|
|
def test_scalar_reference_to_prior_observation(self):
|
|
self.output["observations"].append(self.observation(
|
|
observation_id="obs_2", evidence_id="e2", refers_to="obs_1",
|
|
speaker="Nina", named_person=None, addressee=None,
|
|
))
|
|
validate_observations(self.output, self.case)
|
|
|
|
def test_array_and_invalid_reference_are_rejected(self):
|
|
self.output["observations"][0]["refers_to"] = ["obs_1"]
|
|
with self.assertRaisesRegex(ObservationValidationError, "non-empty string"):
|
|
validate_observations(self.output, self.case)
|
|
self.output["observations"][0]["refers_to"] = "obs_9"
|
|
with self.assertRaisesRegex(ObservationValidationError, "unknown or later"):
|
|
validate_observations(self.output, self.case)
|
|
|
|
def test_qualifier_is_null_or_nonempty_text(self):
|
|
self.output["observations"][0]["qualifier"] = None
|
|
validate_observations(self.output, self.case)
|
|
self.output["observations"][0]["qualifier"] = ""
|
|
with self.assertRaisesRegex(ObservationValidationError, "non-empty string"):
|
|
validate_observations(self.output, self.case)
|
|
|
|
def test_limits_target_must_reference_prior_observation(self):
|
|
self.output["observations"].append(self.observation(
|
|
observation_id="obs_2", evidence_id="e2", refers_to="obs_1",
|
|
limits_target="obs_1", speaker="Nina", named_person=None, addressee=None,
|
|
))
|
|
validate_observations(self.output, self.case)
|
|
self.output["observations"][1]["limits_target"] = "obs_7"
|
|
with self.assertRaisesRegex(ObservationValidationError, "unknown or later"):
|
|
validate_observations(self.output, self.case)
|
|
|
|
def test_multiple_atomic_observations_may_share_evidence(self):
|
|
self.output["observations"].append(self.observation(observation_id="obs_2"))
|
|
validate_observations(self.output, self.case)
|
|
|
|
def test_string_null_is_rejected(self):
|
|
self.output["observations"][0]["named_person"] = "null"
|
|
with self.assertRaisesRegex(ObservationValidationError, "string 'null'"):
|
|
validate_observations(self.output, self.case)
|
|
|
|
def test_malformed_json_is_rejected(self):
|
|
with self.assertRaises(json.JSONDecodeError):
|
|
parse_model_json("{not json")
|
|
|
|
def test_payload_has_exact_live_controls(self):
|
|
payload = build_ollama_payload("qwen3.5:9B", "prompt", 16384, 4096)
|
|
self.assertFalse(payload["think"])
|
|
self.assertFalse(payload["stream"])
|
|
self.assertEqual(payload["options"]["temperature"], 0)
|
|
|
|
def test_fixture_contains_unchanged_a_i_source_evidence(self):
|
|
v1 = load_fixture(Path("tests/gold/evidence_observations_v2/cases.json"))
|
|
original = json.loads(Path("tests/gold/evidence_observations_v1/cases.json").read_text())["cases"]
|
|
self.assertEqual(len(v1), 9)
|
|
self.assertEqual(
|
|
[[item["text"] for item in case["evidence"]] for case in v1],
|
|
[[item["text"] for item in case["evidence"]] for case in original],
|
|
)
|
|
|
|
def test_case_run_preserves_all_artifacts(self):
|
|
raw = json.dumps(self.output, ensure_ascii=False)
|
|
with tempfile.TemporaryDirectory() as temporary:
|
|
root = Path(temporary)
|
|
with patch(
|
|
"src.meeting_lab.evidence_observations_v2.experiment.call_ollama",
|
|
return_value=(raw, {"model": "qwen3.5:9B"}),
|
|
):
|
|
result = run_case(self.case, root, "http://unused", "qwen3.5:9B", 1, 16384, 4096)
|
|
self.assertEqual(result["verdict"], "PASS")
|
|
for filename in (
|
|
"gold_input.json", "gold_expected_observations.json", "prompt.txt",
|
|
"raw_model_response.txt", "parsed_observations.json",
|
|
"validation_result.json", "ollama_metadata.json", "evaluation_result.json",
|
|
):
|
|
self.assertTrue((root / "test_case" / filename).is_file(), filename)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|