From 3229786b5ca3280e505c93812bbd7029d85b8412 Mon Sep 17 00:00:00 2001 From: Martin Date: Thu, 20 Aug 2026 13:39:17 +0200 Subject: [PATCH] Document failed target resolution V0 experiment --- docs/experiments.md | 64 +++++++++++++ scripts/run_target_resolution_experiment.py | 7 ++ .../experiment_target_resolution.py | 92 +++++++++++++++++++ tests/gold/target_resolution_v0/cases.json | 13 +++ tests/test_target_resolution_experiment.py | 75 +++++++++++++++ 5 files changed, 251 insertions(+) create mode 100644 scripts/run_target_resolution_experiment.py create mode 100644 src/meeting_lab/controlled_semantic_derivation/experiment_target_resolution.py create mode 100644 tests/gold/target_resolution_v0/cases.json create mode 100644 tests/test_target_resolution_experiment.py diff --git a/docs/experiments.md b/docs/experiments.md index 3970b88..6a13d86 100644 --- a/docs/experiments.md +++ b/docs/experiments.md @@ -2024,6 +2024,70 @@ ineligible CR-06 path; it is not reliable enough for rejection derivation. Artifacts are preserved under `artifacts/experiments/controlled_rejection_v1/20260820_qwen35_9b_single_run/`. +## EXP-0037 — Target Resolution V0 + +Status: Experimental; FAILED for target resolution + +Date: 2026-08-20 + +Controlled Rejection V1 showed that fine-grained Negative Act Form eligibility +contained false-positive rejection, but its target strategy failed a +self-contained positive and unnecessarily resolved a target for an ineligible +`none` form. This isolated experiment tested target resolution only. It +contains no rejection derivation, status, decision, outcome, responsibility, +topic closure, or production integration. + +Eligibility was deterministic: only `explicit_non_pursuit` could reach the +resolver. TR-05 personal preference, TR-06 recommendation, TR-07 temporary +non-action, and TR-08 `none` stopped before prompt construction and recorded an +explicit skipped-call artifact. This hard gate worked in all four cases. + +The target schema contained exactly `candidate_observation_id`, +`target_observation_id`, and `normalized_target_text`, with local IDs, +same-or-earlier ordering, unique evidence provenance, null consistency, and +recursive normative-field exclusion. TR-01 used the self-contained strategy: +the prompt stated that linkage was deterministically fixed to the candidate and +requested semantic normalization only. TR-02 through TR-04 used paired local +resolution. No original transcript or new Negative Act classification call was +used. + +| Case | Form / eligible | Call | Target result | Verdict | +| --- | --- | --- | --- | --- | +| TR-01 | `explicit_non_pursuit` / yes | yes | Returned string `"null"`; required same-observation target unresolved | FAIL | +| TR-02 | `explicit_non_pursuit` / yes | yes | Returned string `"null"`; `obs_1` unresolved | FAIL | +| TR-03 | `explicit_non_pursuit` / yes | yes | Returned string `"null"`; scoped `obs_1` unresolved | FAIL | +| TR-04 | `explicit_non_pursuit` / yes | yes | Returned string `"null"`; real-plant target unresolved | FAIL | +| TR-05 | `personal_preference` / no | no | Deterministically skipped | PASS | +| TR-06 | `recommendation` / no | no | Deterministically skipped | PASS | +| TR-07 | `temporary_non_action` / no | no | Deterministically skipped | PASS | +| TR-08 | `none` / no | no | Deterministically skipped | PASS | + +Result: four PASS, zero PARTIAL, four FAIL. Exactly four successful Ollama +calls were made, all for eligible cases; there were zero technical call +failures and four structural validation failures. All four raw responses used +the JSON string `"null"` as target ID rather than a supplied observation ID or +JSON null. Wrong-target count and unresolved-target count were therefore four. +No qualifier-preservation claim can be made because no eligible positive target +passed validation. TR-04 alternative isolation likewise could not be +established. No rejection, status, decision, outcome, responsibility, or other +normative leakage occurred, and no rejection derivation was performed. + +Configuration: `qwen3.5:9B`, temperature 0, `think=false`, +`num_ctx=16384`, `num_predict=1024`, no retries, voting, or prompt changes. +Aggregate runner time was 4.034 seconds. + +Conclusion: eligibility gating is successful and should be retained; it fully +prevents unnecessary target calls for ineligible Negative Act forms. Target +Resolution V0 itself failed structurally across all eligible cases. Neither the +self-contained nor paired strategy produced a valid target, and merely +instructing deterministic self-linkage in the semantic prompt did not make the +linkage structurally deterministic. The repeated `"null"` string pattern +requires diagnosis before changing the architecture or prompt. No rejection +derivation is justified by this result. + +Artifacts are preserved under +`artifacts/experiments/target_resolution_v0/20260820_qwen35_9b_single_run/`. + ## EXP-0026 — Topic-oriented Discussion Subject reconstruction V2 prototype Date: 2026-08-11 diff --git a/scripts/run_target_resolution_experiment.py b/scripts/run_target_resolution_experiment.py new file mode 100644 index 0000000..30300f8 --- /dev/null +++ b/scripts/run_target_resolution_experiment.py @@ -0,0 +1,7 @@ +#!/usr/bin/env python3 +import sys +from pathlib import Path +ROOT = Path(__file__).resolve().parents[1] +if str(ROOT) not in sys.path: sys.path.insert(0, str(ROOT)) +from src.meeting_lab.controlled_semantic_derivation.experiment_target_resolution import main +if __name__ == "__main__": raise SystemExit(main()) diff --git a/src/meeting_lab/controlled_semantic_derivation/experiment_target_resolution.py b/src/meeting_lab/controlled_semantic_derivation/experiment_target_resolution.py new file mode 100644 index 0000000..5914022 --- /dev/null +++ b/src/meeting_lab/controlled_semantic_derivation/experiment_target_resolution.py @@ -0,0 +1,92 @@ +#!/usr/bin/env python3 +"""Isolated local target-resolution experiment; performs no rejection derivation.""" +from __future__ import annotations +import argparse, json, time +from pathlib import Path +from typing import Any, Callable +from .experiment_h import DEFAULT_ENDPOINT, DEFAULT_MODEL, DerivationValidationError, OBSERVATION_KEYS, call_ollama +from .experiment_negative_act import validate_classification + +SCHEMA_VERSION="experimental-target-resolution-v0" +TARGET_KEYS={"candidate_observation_id","target_observation_id","normalized_target_text"} +FORBIDDEN={"negative_act_form","rejection_form","explicitly_rejected","status","decision","outcome","topic_status","closed","responsible_person","responsibility","owner","requested_actor","action_item","protocol_category","confidence","relation","relations","graph","unresolved_issue"} +PROMPT="""Resolve and normalize only the concrete action or option referred to by the candidate negative act. Candidate: {candidate}. Strategy: {instruction} Choose only a supplied observation ID. If no unique local target exists, use JSON null for both target fields. Preserve German source meaning, continuation, purpose, location, and other material scope. Do not translate Anlage as asset. Ignore separate positive alternatives. Do not output negative-act form, rejection, status, decision, outcome, responsibility, topic closure, protocol concepts, confidence, relations, or graphs. Return exactly this JSON object with no additional fields: {{"candidate_observation_id":"{candidate}","target_observation_id":"observation ID or null","normalized_target_text":"concise positive action in German or null"}}\nObservations:\n{observations}""" + +def _keys(v:Any, required:set[str], where:str): + if not isinstance(v,dict): raise DerivationValidationError(f"{where} must be an object") + if set(v)!=required: raise DerivationValidationError(f"{where} keys invalid: missing={sorted(required-set(v))}, unknown={sorted(set(v)-required)}") +def _text(v:Any, where:str): + if not isinstance(v,str) or not v.strip(): raise DerivationValidationError(f"{where} must be non-empty") + return v.strip() +def _reject_forbidden(v:Any, where="output"): + if isinstance(v,dict): + bad=FORBIDDEN & set(v) + if bad: raise DerivationValidationError(f"{where} contains forbidden fields: {sorted(bad)}") + for k,x in v.items(): _reject_forbidden(x,f"{where}.{k}") + elif isinstance(v,list): + for i,x in enumerate(v): _reject_forbidden(x,f"{where}[{i}]") +def validate_observations(obs): + if not isinstance(obs,list) or not obs: raise DerivationValidationError("observations must be non-empty") + ids=set(); evidence=set() + for i,o in enumerate(obs): + _keys(o,OBSERVATION_KEYS,f"observations[{i}]"); oid=_text(o["observation_id"],"observation_id"); eid=_text(o["evidence_id"],"evidence_id") + if oid in ids or eid in evidence: raise DerivationValidationError("observation/evidence provenance must be unique") + ids.add(oid); evidence.add(eid); _text(o["content"],"content"); _text(o["speaker"],"speaker") + return ids +def eligibility(negative,obs): + validate_classification(negative,obs) + eligible=negative["negative_act_form"]=="explicit_non_pursuit" + return {"eligible_for_target_resolution":eligible,"reason":None if eligible else "negative_act_form_not_explicit_non_pursuit"} +def validate_target(data,obs,candidate): + ids=validate_observations(obs); _reject_forbidden(data); _keys(data,TARGET_KEYS,"target output") + if _text(data["candidate_observation_id"],"candidate_observation_id")!=candidate: raise DerivationValidationError("candidate observation mismatch") + if candidate not in ids: raise DerivationValidationError("unknown candidate observation") + target=data["target_observation_id"]; normalized=data["normalized_target_text"] + if target is None: + if normalized is not None: raise DerivationValidationError("null target requires null text") + else: + target=_text(target,"target_observation_id") + if target not in ids: raise DerivationValidationError("unknown target observation") + if [o["observation_id"] for o in obs].index(target)>[o["observation_id"] for o in obs].index(candidate): raise DerivationValidationError("target must not occur after candidate") + _text(normalized,"normalized_target_text") + return data +def build_prompt(case): + gate=eligibility(case["negative_act"],case["observations"]) + if not gate["eligible_for_target_resolution"]: raise DerivationValidationError("ineligible case must not build a target prompt") + candidate=case["negative_act"]["observation_id"] + instruction=(f"The target linkage is deterministically fixed to {candidate}; output that exact target ID and only normalize its positive action meaning." if case["strategy"]=="self_contained" else "Resolve the unique preceding local observation that supplies the referenced action.") + return PROMPT.format(candidate=candidate,instruction=instruction,observations=json.dumps(case["observations"],ensure_ascii=False,indent=2)) +def _concepts(text,groups): + folded=(text or "").casefold(); return all(any(alias.casefold() in folded for alias in group) for group in groups) +def evaluate(case,gate,called,target): + e=case["expected"]; eligible=gate["eligible_for_target_resolution"]==e["eligible"]; call_ok=called==e["eligible"] + if not e["eligible"]: + label="PASS" if eligible and call_ok and target is None else "FAIL" + return {"case_id":case["case_id"],"classification":label,"negative_act_form":case["negative_act"]["negative_act_form"],"eligible":gate["eligible_for_target_resolution"],"target_resolution_call_made":called,"expected_target_observation_id":None,"actual_target_observation_id":None,"normalized_target_text":None,"material_scope_preserved":True,"alternative_isolation":True,"normative_leakage":False} + text=target["normalized_target_text"]; target_ok=target["target_observation_id"]==e["target_observation_id"]; concepts=_concepts(text,e["concepts"]); material=_concepts(text,e["material_concepts"]); isolated=not any(x.casefold() in (text or "").casefold() for x in e["forbidden_concepts"]) + label="PASS" if eligible and call_ok and target_ok and concepts and material and isolated else ("PARTIAL" if eligible and call_ok and target_ok and material and isolated else "FAIL") + return {"case_id":case["case_id"],"classification":label,"negative_act_form":case["negative_act"]["negative_act_form"],"eligible":gate["eligible_for_target_resolution"],"target_resolution_call_made":called,"expected_target_observation_id":e["target_observation_id"],"actual_target_observation_id":target["target_observation_id"],"normalized_target_text":text,"normalized_action_correct":concepts,"material_scope_preserved":material,"alternative_isolation":isolated,"normative_leakage":False} +def load_cases(path): + data=json.loads(path.read_text(encoding="utf-8")); _keys(data,{"schema_version","cases"},"fixture") + if data["schema_version"]!=SCHEMA_VERSION: raise DerivationValidationError("unexpected schema version") + for case in data["cases"]: validate_observations(case["observations"]); validate_classification(case["negative_act"],case["observations"]) + return data["cases"] +def _write(path,value): path.write_text(json.dumps(value,ensure_ascii=False,indent=2)+"\n",encoding="utf-8") +def run(args, resolver:Callable=call_ollama): + cases=load_cases(args.cases); args.output.mkdir(parents=True,exist_ok=False); _write(args.output/"gold_cases.json",{"schema_version":SCHEMA_VERSION,"cases":cases}) + evaluations=[]; calls=technical_failures=structural_failures=0; started=time.perf_counter() + for case in cases: + folder=args.output/case["case_id"].lower(); folder.mkdir(); _write(folder/"v3_style_input_observations.json",case["observations"]); _write(folder/"negative_act_form.json",case["negative_act"]) + gate=eligibility(case["negative_act"],case["observations"]); _write(folder/"eligibility.json",gate) + if not gate["eligible_for_target_resolution"]: + skipped={"call_made":False,"reason":gate["reason"]}; _write(folder/"target_resolution_skipped.json",skipped); ev=evaluate(case,gate,False,None) + else: + prompt=build_prompt(case); (folder/"prompt.txt").write_text(prompt,encoding="utf-8") + try: + raw,meta=resolver(args.endpoint,args.model,prompt,args.timeout,args.num_ctx,args.num_predict); calls+=1; (folder/"raw_model_response.txt").write_text(raw+"\n",encoding="utf-8"); _write(folder/"ollama_metadata.json",meta); parsed=json.loads(raw); _write(folder/"parsed_target_resolution.json",parsed); validate_target(parsed,case["observations"],case["negative_act"]["observation_id"]); _write(folder/"structural_validation.json",{"valid":True}); ev=evaluate(case,gate,True,parsed) + except Exception as exc: + structural_failures+=1; _write(folder/"structural_validation.json",{"valid":False,"error":str(exc)}); ev={"case_id":case["case_id"],"classification":"FAIL","negative_act_form":case["negative_act"]["negative_act_form"],"eligible":True,"target_resolution_call_made":True,"error":str(exc)} + _write(folder/"evaluation.json",ev); evaluations.append(ev) + summary={"experiment":"target_resolution_v0","model":args.model,"target_resolution_llm_call_count":calls,"technical_failed_call_count":technical_failures,"structural_validation_failure_count":structural_failures,"runtime_seconds":round(time.perf_counter()-started,3),"counts":{x:sum(e["classification"]==x for e in evaluations) for x in ["PASS","PARTIAL","FAIL"]},"evaluations":evaluations}; _write(args.output/"summary.json",summary); return summary +def main(): + p=argparse.ArgumentParser(); p.add_argument("cases",type=Path); p.add_argument("-o","--output",type=Path,required=True); p.add_argument("--model",default=DEFAULT_MODEL); p.add_argument("--endpoint",default=DEFAULT_ENDPOINT); p.add_argument("--timeout",type=int,default=300); p.add_argument("--num-ctx",type=int,default=16384); p.add_argument("--num-predict",type=int,default=1024); print(json.dumps(run(p.parse_args()),ensure_ascii=False,indent=2)); return 0 diff --git a/tests/gold/target_resolution_v0/cases.json b/tests/gold/target_resolution_v0/cases.json new file mode 100644 index 0000000..6a1b405 --- /dev/null +++ b/tests/gold/target_resolution_v0/cases.json @@ -0,0 +1,13 @@ +{ + "schema_version": "experimental-target-resolution-v0", + "cases": [ + {"case_id":"TR-01","description":"self-contained continuation target","strategy":"self_contained","observations":[{"observation_id":"obs_1","evidence_id":"e1","content":"Martin: Mit Dr. Schlummer arbeiten wir nicht weiter.","speaker":"Martin","named_person":"Dr. Schlummer","addressee":null}],"negative_act":{"observation_id":"obs_1","negative_act_form":"explicit_non_pursuit","normalized_action_text":"working with Dr. Schlummer"},"expected":{"eligible":true,"target_observation_id":"obs_1","concepts":[["Zusammenarbeit","arbeiten"],["Schlummer"],["fortsetzen","weiter"]],"material_concepts":[],"forbidden_concepts":[]}}, + {"case_id":"TR-02","description":"paired pronoun target","strategy":"paired","observations":[{"observation_id":"obs_1","evidence_id":"e1","content":"Martin: Eine Möglichkeit wäre, die externe Lösung weiterzuverfolgen.","speaker":"Martin","named_person":null,"addressee":null},{"observation_id":"obs_2","evidence_id":"e2","content":"Martin: Das verfolgen wir nicht weiter.","speaker":"Martin","named_person":null,"addressee":null}],"negative_act":{"observation_id":"obs_2","negative_act_form":"explicit_non_pursuit","normalized_action_text":"verfolgen wir nicht weiter"},"expected":{"eligible":true,"target_observation_id":"obs_1","concepts":[["externe Lösung"],["weiterverfolgen","weiter verfolgen"]],"material_concepts":[],"forbidden_concepts":[]}}, + {"case_id":"TR-03","description":"scoped location and purpose target","strategy":"paired","observations":[{"observation_id":"obs_1","evidence_id":"e1","content":"Martin: Für den Druckversuch steht die reale Anlage zur Diskussion.","speaker":"Martin","named_person":null,"addressee":null},{"observation_id":"obs_2","evidence_id":"e2","content":"Martin: Die reale Anlage nutzen wir dafür nicht.","speaker":"Martin","named_person":null,"addressee":null}],"negative_act":{"observation_id":"obs_2","negative_act_form":"explicit_non_pursuit","normalized_action_text":"reale Anlage dafür nicht nutzen"},"expected":{"eligible":true,"target_observation_id":"obs_1","concepts":[["Anlage"],["nutzen"]],"material_concepts":[["real"],["Druckversuch"]],"forbidden_concepts":[]}}, + {"case_id":"TR-04","description":"rejection plus alternative","strategy":"paired","observations":[{"observation_id":"obs_1","evidence_id":"e1","content":"Martin: Wir könnten den Versuch in der realen Anlage durchführen.","speaker":"Martin","named_person":null,"addressee":null},{"observation_id":"obs_2","evidence_id":"e2","content":"Martin: Das machen wir nicht; wir testen stattdessen im Technikum.","speaker":"Martin","named_person":null,"addressee":null}],"negative_act":{"observation_id":"obs_2","negative_act_form":"explicit_non_pursuit","normalized_action_text":"Versuch in der realen Anlage nicht durchführen"},"expected":{"eligible":true,"target_observation_id":"obs_1","concepts":[["Versuch"],["durchführen"]],"material_concepts":[["real"],["Anlage"]],"forbidden_concepts":["Technikum"]}}, + {"case_id":"TR-05","description":"personal preference","strategy":"paired","observations":[{"observation_id":"obs_1","evidence_id":"e1","content":"Martin: Wir könnten die reale Anlage für den Versuch nutzen.","speaker":"Martin","named_person":null,"addressee":null},{"observation_id":"obs_2","evidence_id":"e2","content":"Martin: Ich würde das nicht machen.","speaker":"Martin","named_person":null,"addressee":null}],"negative_act":{"observation_id":"obs_2","negative_act_form":"personal_preference","normalized_action_text":"Ich würde das nicht machen"},"expected":{"eligible":false,"target_observation_id":null,"concepts":[],"material_concepts":[],"forbidden_concepts":[]}}, + {"case_id":"TR-06","description":"recommendation","strategy":"paired","observations":[{"observation_id":"obs_1","evidence_id":"e1","content":"Martin: Wir könnten die reale Anlage verwenden.","speaker":"Martin","named_person":null,"addressee":null},{"observation_id":"obs_2","evidence_id":"e2","content":"Martin: Ich würde eher davon abraten.","speaker":"Martin","named_person":null,"addressee":null}],"negative_act":{"observation_id":"obs_2","negative_act_form":"recommendation","normalized_action_text":"advise against using the real asset"},"expected":{"eligible":false,"target_observation_id":null,"concepts":[],"material_concepts":[],"forbidden_concepts":[]}}, + {"case_id":"TR-07","description":"temporary non-action","strategy":"paired","observations":[{"observation_id":"obs_1","evidence_id":"e1","content":"Martin: Wir könnten die Waschstufe einbauen.","speaker":"Martin","named_person":null,"addressee":null},{"observation_id":"obs_2","evidence_id":"e2","content":"Martin: Das machen wir erstmal noch nicht.","speaker":"Martin","named_person":null,"addressee":null}],"negative_act":{"observation_id":"obs_2","negative_act_form":"temporary_non_action","normalized_action_text":"install the washing stage"},"expected":{"eligible":false,"target_observation_id":null,"concepts":[],"material_concepts":[],"forbidden_concepts":[]}}, + {"case_id":"TR-08","description":"concern only","strategy":"paired","observations":[{"observation_id":"obs_1","evidence_id":"e1","content":"Martin: Wir könnten das neue Material einsetzen.","speaker":"Martin","named_person":null,"addressee":null},{"observation_id":"obs_2","evidence_id":"e2","content":"Martin: Das wäre kritisch.","speaker":"Martin","named_person":null,"addressee":null}],"negative_act":{"observation_id":"obs_2","negative_act_form":"none","normalized_action_text":null},"expected":{"eligible":false,"target_observation_id":null,"concepts":[],"material_concepts":[],"forbidden_concepts":[]}} + ] +} diff --git a/tests/test_target_resolution_experiment.py b/tests/test_target_resolution_experiment.py new file mode 100644 index 0000000..011a4d0 --- /dev/null +++ b/tests/test_target_resolution_experiment.py @@ -0,0 +1,75 @@ +import argparse, copy, json, tempfile, unittest +from pathlib import Path +from unittest.mock import Mock + +from src.meeting_lab.controlled_semantic_derivation.experiment_h import DerivationValidationError +import src.meeting_lab.controlled_semantic_derivation.experiment_target_resolution as module + +GOLD=Path("tests/gold/target_resolution_v0/cases.json") +CASES=module.load_cases(GOLD); BY_ID={c["case_id"]:c for c in CASES} + +def target(case, target_id=None, text="konkrete Zielhandlung"): + return {"candidate_observation_id":case["negative_act"]["observation_id"],"target_observation_id":target_id if target_id is not None else case["expected"]["target_observation_id"],"normalized_target_text":text} + +class TargetResolutionTests(unittest.TestCase): + def test_eligibility_enum_boundary(self): + for cid in ("TR-01","TR-02","TR-03","TR-04"): + self.assertTrue(module.eligibility(BY_ID[cid]["negative_act"],BY_ID[cid]["observations"])["eligible_for_target_resolution"]) + for cid in ("TR-05","TR-06","TR-07","TR-08"): + gate=module.eligibility(BY_ID[cid]["negative_act"],BY_ID[cid]["observations"]) + self.assertFalse(gate["eligible_for_target_resolution"]); self.assertEqual(gate["reason"],"negative_act_form_not_explicit_non_pursuit") + + def test_ineligible_cases_never_call_resolver_and_record_skip(self): + fixture={"schema_version":module.SCHEMA_VERSION,"cases":[BY_ID[x] for x in ("TR-05","TR-06","TR-07","TR-08")]}; resolver=Mock() + with tempfile.TemporaryDirectory() as tmp: + root=Path(tmp); cases=root/"cases.json"; cases.write_text(json.dumps(fixture)); out=root/"out" + summary=module.run(argparse.Namespace(cases=cases,output=out,endpoint="x",model="qwen3.5:9B",timeout=1,num_ctx=16384,num_predict=1024),resolver) + self.assertEqual(summary["target_resolution_llm_call_count"],0); resolver.assert_not_called() + for cid in ("tr-05","tr-06","tr-07","tr-08"): + skipped=json.loads((out/cid/"target_resolution_skipped.json").read_text()); self.assertFalse(skipped["call_made"]) + + def test_self_contained_target_equals_candidate(self): + c=BY_ID["TR-01"]; self.assertEqual(module.validate_target(target(c,text="Zusammenarbeit mit Dr. Schlummer fortsetzen"),c["observations"],"obs_1")["target_observation_id"],"obs_1") + + def test_paired_target_precedes_candidate(self): + c=BY_ID["TR-02"]; self.assertEqual(module.validate_target(target(c,text="externe Lösung weiterverfolgen"),c["observations"],"obs_2")["target_observation_id"],"obs_1") + + def test_target_after_candidate_rejected(self): + c=copy.deepcopy(BY_ID["TR-02"]); data={"candidate_observation_id":"obs_1","target_observation_id":"obs_2","normalized_target_text":"x"} + with self.assertRaises(DerivationValidationError): module.validate_target(data,c["observations"],"obs_1") + + def test_unknown_candidate_and_target_rejected(self): + c=BY_ID["TR-02"] + with self.assertRaises(DerivationValidationError): module.validate_target({"candidate_observation_id":"missing","target_observation_id":"obs_1","normalized_target_text":"x"},c["observations"],"missing") + with self.assertRaises(DerivationValidationError): module.validate_target({"candidate_observation_id":"obs_2","target_observation_id":"missing","normalized_target_text":"x"},c["observations"],"obs_2") + + def test_duplicate_observation_and_evidence_ids_rejected(self): + for field in ("observation_id","evidence_id"): + obs=copy.deepcopy(BY_ID["TR-02"]["observations"]); obs[1][field]=obs[0][field] + with self.assertRaises(DerivationValidationError): module.validate_observations(obs) + + def test_null_and_non_null_text_constraints(self): + c=BY_ID["TR-02"] + valid={"candidate_observation_id":"obs_2","target_observation_id":None,"normalized_target_text":None}; self.assertEqual(module.validate_target(valid,c["observations"],"obs_2"),valid) + for bad in ({"candidate_observation_id":"obs_2","target_observation_id":None,"normalized_target_text":"x"},{"candidate_observation_id":"obs_2","target_observation_id":"obs_1","normalized_target_text":""}): + with self.assertRaises(DerivationValidationError): module.validate_target(bad,c["observations"],"obs_2") + + def test_unknown_and_recursive_forbidden_fields_rejected(self): + c=BY_ID["TR-02"] + for extra in ({"extra":1},{"nested":{"status":"rejected"}}): + data=target(c,text="x"); data.update(extra) + with self.assertRaises(DerivationValidationError): module.validate_target(data,c["observations"],"obs_2") + + def test_self_contained_prompt_fixes_linkage_deterministically(self): + prompt=module.build_prompt(BY_ID["TR-01"]); self.assertIn("deterministically fixed to obs_1",prompt); self.assertIn("only normalize",prompt) + + def test_ineligible_prompt_is_impossible(self): + with self.assertRaises(DerivationValidationError): module.build_prompt(BY_ID["TR-05"]) + + def test_experiment_has_no_rejection_derivation(self): + self.assertFalse(hasattr(module,"derive")); self.assertNotIn("explicitly_rejected",module.TARGET_KEYS); self.assertNotIn("status",module.TARGET_KEYS) + + def test_evaluation_preserves_scope_and_alternative_contract(self): + c=BY_ID["TR-04"]; gate=module.eligibility(c["negative_act"],c["observations"]); result=module.evaluate(c,gate,True,target(c,text="Versuch in der realen Anlage durchführen")); self.assertEqual(result["classification"],"PASS"); self.assertTrue(result["alternative_isolation"]) + +if __name__=="__main__": unittest.main()