Add evidence-near semantic architecture experiments

Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
This commit is contained in:
2026-08-19 15:46:22 +02:00
parent bcb197a908
commit 18beb3385f
29 changed files with 4542 additions and 0 deletions
@@ -0,0 +1,168 @@
import json
import tempfile
import unittest
from copy import deepcopy
from pathlib import Path
from unittest.mock import patch
from src.meeting_lab.evidence_observations.experiment import (
SCHEMA_VERSION,
ObservationValidationError,
build_ollama_payload,
load_fixture,
parse_model_json,
run_case,
validate_observations,
)
class EvidenceObservationExperimentTests(unittest.TestCase):
def setUp(self) -> None:
self.case = {
"case_id": "test_case",
"description": "Validator fixture.",
"subject_id": "subject_test",
"subject": "Prüfung der Messdaten",
"evidence": [
{"evidence_id": "e1", "text": "Nina, prüfst du die Daten?"},
{"evidence_id": "e2", "text": "Ja, ich prüfe sie."},
],
"expected_observations": [],
}
self.output = {
"schema_version": SCHEMA_VERSION,
"subject_id": "subject_test",
"subject": "Prüfung der Messdaten",
"observations": [self.observation()],
}
self.case["expected_observations"] = deepcopy(self.output["observations"])
def observation(self, **updates):
value = {
"observation_id": "obs_1",
"evidence_id": "e1",
"content": "Nina wird um Prüfung gebeten.",
"target": "discussion_subject",
"relation": "none",
"modality": "interpersonal_request",
"temporality": "future",
"evaluation": "none",
"agreement": "none",
"responsibility": "named",
"person": "Nina",
"uncertainty": "absent",
"clarification_need": "none",
"scope": "absent",
}
value.update(updates)
return value
def test_valid_observation_and_discussion_subject_target(self):
self.assertIs(validate_observations(self.output, self.case), self.output)
def test_multiple_observations_from_one_evidence_unit_and_observation_target(self):
second = self.observation(
observation_id="obs_2", target="obs_1", relation="supports"
)
self.output["observations"].append(second)
validate_observations(self.output, self.case)
def test_plural_target_is_allowed_for_joint_reference(self):
self.output["observations"].extend(
[
self.observation(observation_id="obs_2"),
self.observation(
observation_id="obs_3",
target=["obs_1", "obs_2"],
relation="qualifies",
),
]
)
validate_observations(self.output, self.case)
def test_unknown_evidence_reference_is_rejected(self):
self.output["observations"][0]["evidence_id"] = "missing"
with self.assertRaisesRegex(ObservationValidationError, "unknown evidence"):
validate_observations(self.output, self.case)
def test_unknown_observation_target_is_rejected(self):
self.output["observations"][0]["target"] = "obs_9"
with self.assertRaisesRegex(ObservationValidationError, "unknown or later"):
validate_observations(self.output, self.case)
def test_invalid_relation_is_rejected(self):
self.output["observations"][0]["relation"] = "causes"
with self.assertRaisesRegex(ObservationValidationError, "relation is invalid"):
validate_observations(self.output, self.case)
def test_invalid_modality_is_rejected(self):
self.output["observations"][0]["modality"] = "proposal"
with self.assertRaisesRegex(ObservationValidationError, "modality is invalid"):
validate_observations(self.output, self.case)
def test_invalid_responsibility_person_combinations_are_rejected(self):
self.output["observations"][0].update(responsibility="none", person="Nina")
with self.assertRaisesRegex(ObservationValidationError, "person must be JSON null"):
validate_observations(self.output, self.case)
self.output["observations"][0].update(responsibility="accepted", person=None)
with self.assertRaisesRegex(ObservationValidationError, "person must be a non-empty"):
validate_observations(self.output, self.case)
def test_scope_uses_absent_or_nonempty_evidence_grounded_text(self):
validate_observations(self.output, self.case)
self.output["observations"][0]["scope"] = "bis Freitag"
validate_observations(self.output, self.case)
self.output["observations"][0]["scope"] = None
with self.assertRaisesRegex(ObservationValidationError, "non-empty string"):
validate_observations(self.output, self.case)
def test_string_null_is_rejected_in_text_fields(self):
self.output["observations"][0]["scope"] = "null"
with self.assertRaisesRegex(ObservationValidationError, "string 'null'"):
validate_observations(self.output, self.case)
def test_malformed_model_json_is_rejected(self):
with self.assertRaises(json.JSONDecodeError):
parse_model_json("{not json")
def test_payload_has_exact_live_controls(self):
payload = build_ollama_payload("qwen3.5:9B", "prompt", 16384, 4096)
self.assertIs(payload["think"], False)
self.assertIs(payload["stream"], False)
self.assertEqual(payload["format"], "json")
self.assertEqual(payload["options"]["temperature"], 0)
def test_fixture_contains_all_nine_cases(self):
cases = load_fixture(Path("tests/gold/evidence_observations_v1/cases.json"))
self.assertEqual(len(cases), 9)
self.assertEqual(cases[0]["case_id"], "a_idea_only")
self.assertEqual(cases[-1]["case_id"], "i_outcome_and_unresolved")
def test_case_run_preserves_all_artifacts(self):
raw = json.dumps(self.output, ensure_ascii=False)
with tempfile.TemporaryDirectory() as temporary:
root = Path(temporary)
with patch(
"src.meeting_lab.evidence_observations.experiment.call_ollama",
return_value=(raw, {"model": "qwen3.5:9B"}),
):
result = run_case(
self.case, root, "http://unused", "qwen3.5:9B", 1, 16384, 4096
)
self.assertEqual(result["verdict"], "PASS")
for filename in (
"gold_input.json",
"gold_expected_observations.json",
"prompt.txt",
"raw_model_response.txt",
"parsed_observations.json",
"validation_result.json",
"ollama_metadata.json",
"evaluation_result.json",
):
self.assertTrue((root / "test_case" / filename).is_file(), filename)
if __name__ == "__main__":
unittest.main()