diff --git a/docs/experiments.md b/docs/experiments.md index d6cf429..16c1b96 100644 --- a/docs/experiments.md +++ b/docs/experiments.md @@ -1890,6 +1890,78 @@ cross-pattern reconciliation is justified. Artifacts are preserved under `artifacts/experiments/explicit_rejection_gold_v0/20260820_qwen35_9b_single_run/`. +## EXP-0035 — Negative Act Form V0 + +Status: Experimental; successful for form classification with normalization +limitations + +Date: 2026-08-20 + +EXP-0034 failed because the binary `explicit_action_rejection | none` question +collapsed materially different negative acts. It missed self-contained +non-pursuit and promoted personal preference, recommendation and temporary +non-action to rejection. This isolated follow-up tested only whether those +evidence-near forms can be distinguished before any normative derivation. It +does not derive rejection, decision, outcome, topic closure, responsibility or +protocol status, and EXP-0034 remained unchanged. + +The strict output schema contains exactly `observation_id`, +`negative_act_form` and `normalized_action_text`. The closed form vocabulary is +`explicit_non_pursuit`, `personal_preference`, `recommendation`, +`temporary_non_action` and `none`. Non-`none` forms require non-empty normalized +action text; `none` requires null. Rejection, status, decision, outcome, +responsibility and other normative fields are forbidden recursively. Local +context may resolve a candidate observation's pronoun, but the schema contains +no target relation and the experiment exposes no derivation function. + +Gold results: + +- NA-01 explicit non-pursuit: PARTIAL. The form was correct; `working with Dr. + Schlummer` omitted the continuation aspect from normalization. +- NA-02 paraphrased explicit non-pursuit: PASS. +- NA-03 personal preference: PARTIAL. The form was correct, but normalization + repeated `Ich würde das nicht machen` instead of resolving the real-plant + trial target. +- NA-04 negative recommendation: PARTIAL. The form was correct; the normalized + English action used the loose rendering `real asset` for `reale Anlage`. +- NA-05 temporary non-action: PASS. +- NA-06 concern only: PASS with `none` and null action text. +- NA-07 uncertainty: PASS with `none` and null action text. +- NA-08 factual negation: PASS with `none` and null action text. + +Expected-versus-actual form confusion was entirely diagonal: + +| Expected form | Actual form | Count | +| --- | --- | ---: | +| `explicit_non_pursuit` | `explicit_non_pursuit` | 2 | +| `personal_preference` | `personal_preference` | 1 | +| `recommendation` | `recommendation` | 1 | +| `temporary_non_action` | `temporary_non_action` | 1 | +| `none` | `none` | 3 | + +Configuration: exactly eight successful sequential `qwen3.5:9B` calls, one +per case, temperature 0, `think=false`, `num_ctx=16384`, +`num_predict=1024`, no retries, no voting and no prompt changes. There were zero +technical failures. Aggregate runner time was 8.688 seconds; summed per-call +time was 8.686 seconds, with 4,183 prompt-evaluation tokens and 310 evaluation +tokens. + +The result was five PASS, three PARTIAL and zero FAIL. All eight +`negative_act_form` classifications matched Gold. There was no unsupported +semantic strengthening and no rejection, status, decision, outcome, +responsibility or topic-closure leakage. Normalized action meaning was fully +acceptable in five cases and imperfect in three. + +Conclusion: the finer evidence-near form vocabulary successfully distinguished +the four semantic boundaries that defeated the binary rejection experiment in +this small Gold set. The result supports separating negative-act-form +recognition from later normative derivation, but local target normalization is +not yet uniformly reliable. It does not justify modifying EXP-0034, deriving +rejection, production integration or beginning cross-pattern reconciliation. + +Artifacts are preserved under +`artifacts/experiments/negative_act_form_v0/20260820_qwen35_9b_single_run/`. + ## EXP-0026 — Topic-oriented Discussion Subject reconstruction V2 prototype Date: 2026-08-11 diff --git a/scripts/run_negative_act_form_experiment.py b/scripts/run_negative_act_form_experiment.py new file mode 100644 index 0000000..a1b2bf8 --- /dev/null +++ b/scripts/run_negative_act_form_experiment.py @@ -0,0 +1,16 @@ +#!/usr/bin/env python3 +"""Repository entry point for the Negative Act Form experiment.""" + +import sys +from pathlib import Path + + +REPO_ROOT = Path(__file__).resolve().parents[1] +if str(REPO_ROOT) not in sys.path: + sys.path.insert(0, str(REPO_ROOT)) + +from src.meeting_lab.controlled_semantic_derivation.experiment_negative_act import main # noqa: E402 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/meeting_lab/controlled_semantic_derivation/experiment_negative_act.py b/src/meeting_lab/controlled_semantic_derivation/experiment_negative_act.py new file mode 100644 index 0000000..3132d29 --- /dev/null +++ b/src/meeting_lab/controlled_semantic_derivation/experiment_negative_act.py @@ -0,0 +1,277 @@ +#!/usr/bin/env python3 +"""Isolated evidence-near Negative Act Form classification experiment.""" + +from __future__ import annotations + +import argparse +import json +import time +from pathlib import Path +from typing import Any + +from .experiment_h import ( + DEFAULT_ENDPOINT, + DEFAULT_MODEL, + DerivationValidationError, + OBSERVATION_KEYS, + build_ollama_payload, + call_ollama, +) + + +GOLD_SCHEMA_VERSION = "experimental-negative-act-form-gold-v0" +RECOGNITION_KEYS = {"observation_id", "negative_act_form", "normalized_action_text"} +NEGATIVE_ACT_FORMS = { + "explicit_non_pursuit", "personal_preference", "recommendation", + "temporary_non_action", "none", +} +FORBIDDEN_LLM_KEYS = { + "rejection_form", "explicitly_rejected", "status", "decision", "outcome", + "topic_status", "responsible_person", "responsibility", "owner", + "requested_actor", "action_item", "protocol", "protocol_category", + "confidence", "relation", "relations", "graph", "unresolved_issue", +} + +PROMPT_TEMPLATE = """Classify only the negative semantic form expressed by the candidate observation, using earlier supplied V3-style observations only as local context for pronouns or shortened references. + +The candidate observation is {candidate_observation_id}. + +Choose exactly one negative_act_form: +- explicit_non_pursuit: explicitly states that an action, option, collaboration, or course will not be continued or pursued. This is stronger than preference, advice, or temporary delay. +- personal_preference: the speaker states what they personally would or would not do, without establishing collective non-pursuit. +- recommendation: the speaker advises for or against an action without establishing abandonment. +- temporary_non_action: the action is postponed, deferred, or explicitly not done for now without abandonment. +- none: none of those four forms is present, including mere concern, uncertainty, negative sentiment, or factual negation. + +Do not collapse non-pursuit into temporary non-action. Do not convert a personal conditional preference into collective non-pursuit. Do not convert advice into non-pursuit. Speaker identity does not change personal preference into collective non-pursuit. + +When the form is not none, return concise normalized action meaning. Resolve a pronoun only from the supplied local context. If its target is genuinely ambiguous, return none rather than guessing. When the form is none, normalized_action_text must be null. Keep normalized action text in the observation language. + +Do not derive or output rejection, status, decision, outcome, topic closure, responsibility, ownership, Action Item, protocol category, confidence, relations, graphs, or unresolved issues. + +Return exactly this JSON shape and no additional fields: +{{ + "observation_id": "{candidate_observation_id}", + "negative_act_form": "explicit_non_pursuit | personal_preference | recommendation | temporary_non_action | none", + "normalized_action_text": "concise action meaning" | null +}} + +V3-style observations: +{observations_json} +""" + + +def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None: + missing = required - value.keys() + unknown = value.keys() - required + if missing: + raise DerivationValidationError(f"{location} missing required keys: {sorted(missing)}") + if unknown: + raise DerivationValidationError(f"{location} has unknown keys: {sorted(unknown)}") + + +def _nonempty_text(value: Any, location: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise DerivationValidationError(f"{location} must be a non-empty string") + return value.strip() + + +def _validate_observations(observations: Any) -> None: + if not isinstance(observations, list) or not observations: + raise DerivationValidationError("observations must be a non-empty list") + seen_observations: set[str] = set() + seen_evidence: set[str] = set() + for index, observation in enumerate(observations): + location = f"observations[{index}]" + if not isinstance(observation, dict): + raise DerivationValidationError(f"{location} must be an object") + _exact_keys(observation, OBSERVATION_KEYS, location) + observation_id = _nonempty_text(observation["observation_id"], f"{location}.observation_id") + evidence_id = _nonempty_text(observation["evidence_id"], f"{location}.evidence_id") + if observation_id in seen_observations or evidence_id in seen_evidence: + raise DerivationValidationError("observation and evidence provenance must be unique") + seen_observations.add(observation_id) + seen_evidence.add(evidence_id) + _nonempty_text(observation["content"], f"{location}.content") + _nonempty_text(observation["speaker"], f"{location}.speaker") + for field in ("named_person", "addressee"): + if observation[field] is not None: + _nonempty_text(observation[field], f"{location}.{field}") + + +def load_gold_cases(path: Path) -> list[dict[str, Any]]: + data = json.loads(path.read_text(encoding="utf-8-sig")) + if not isinstance(data, dict): + raise DerivationValidationError("Gold fixture must be an object") + _exact_keys(data, {"schema_version", "cases"}, "Gold fixture") + if data["schema_version"] != GOLD_SCHEMA_VERSION: + raise DerivationValidationError("unexpected Gold fixture schema_version") + cases = data["cases"] + if not isinstance(cases, list) or not cases: + raise DerivationValidationError("Gold fixture cases must be a non-empty list") + seen: set[str] = set() + for case in cases: + _exact_keys(case, {"case_id", "description", "observations", "expected"}, "Gold case") + case_id = _nonempty_text(case["case_id"], "Gold case.case_id") + if case_id in seen: + raise DerivationValidationError(f"duplicate case ID: {case_id}") + seen.add(case_id) + _validate_observations(case["observations"]) + if len(case["observations"]) not in (1, 2): + raise DerivationValidationError("Negative Act cases require one or two observations") + return cases + + +def build_prompt(case: dict[str, Any]) -> str: + observations = case["observations"] + _validate_observations(observations) + candidate_id = observations[-1]["observation_id"] + return PROMPT_TEMPLATE.format( + candidate_observation_id=candidate_id, + observations_json=json.dumps(observations, ensure_ascii=False, indent=2), + ) + + +def parse_model_json(raw_text: str) -> dict[str, Any]: + data = json.loads(raw_text) + if not isinstance(data, dict): + raise DerivationValidationError("semantic classification must be an object") + return data + + +def _reject_forbidden_keys(value: Any, location: str = "output") -> None: + if isinstance(value, dict): + forbidden = FORBIDDEN_LLM_KEYS.intersection(value) + if forbidden: + raise DerivationValidationError(f"{location} contains forbidden semantic keys: {sorted(forbidden)}") + for key, item in value.items(): + _reject_forbidden_keys(item, f"{location}.{key}") + elif isinstance(value, list): + for index, item in enumerate(value): + _reject_forbidden_keys(item, f"{location}[{index}]") + + +def validate_classification(data: Any, observations: list[dict[str, Any]]) -> dict[str, Any]: + _validate_observations(observations) + if not isinstance(data, dict): + raise DerivationValidationError("semantic classification must be an object") + _reject_forbidden_keys(data) + _exact_keys(data, RECOGNITION_KEYS, "output") + observation_id = _nonempty_text(data["observation_id"], "output.observation_id") + if observation_id not in {item["observation_id"] for item in observations}: + raise DerivationValidationError("classification references unknown observation") + form = data["negative_act_form"] + if form not in NEGATIVE_ACT_FORMS: + raise DerivationValidationError("negative_act_form has an unsupported value") + action_text = data["normalized_action_text"] + if form == "none": + if action_text is not None: + raise DerivationValidationError("none form requires null normalized_action_text") + else: + _nonempty_text(action_text, "output.normalized_action_text") + return data + + +def _concepts_present(text: str | None, concepts: list[list[str]]) -> bool: + if not concepts: + return text is None + if not isinstance(text, str): + return False + folded = text.casefold() + return all(any(alias.casefold() in folded for alias in alternatives) for alternatives in concepts) + + +def evaluate_case(case: dict[str, Any], classification: dict[str, Any]) -> dict[str, Any]: + validate_classification(classification, case["observations"]) + expected = case["expected"] + observation_correct = classification["observation_id"] == expected["observation_id"] + form_correct = classification["negative_act_form"] == expected["negative_act_form"] + action_correct = _concepts_present(classification["normalized_action_text"], expected["action_concepts"]) + unsupported_strengthening = expected["negative_act_form"] == "none" and classification["negative_act_form"] != "none" + classification_label = "PASS" if observation_correct and form_correct and action_correct else ("PARTIAL" if observation_correct and form_correct else "FAIL") + return { + "case_id": case["case_id"], "classification": classification_label, + "expected_negative_act_form": expected["negative_act_form"], + "actual_negative_act_form": classification["negative_act_form"], + "observation_id_correct": observation_correct, + "normalized_action_meaning_correct": action_correct, + "unsupported_semantic_strengthening": unsupported_strengthening, + "normative_leakage": False, + } + + +def _write_json(path: Path, value: Any) -> None: + path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + + +def run_experiment(args: argparse.Namespace) -> dict[str, Any]: + cases = load_gold_cases(args.cases) + args.output.mkdir(parents=True, exist_ok=False) + _write_json(args.output / "gold_cases.json", {"schema_version": GOLD_SCHEMA_VERSION, "cases": cases}) + evaluations: list[dict[str, Any]] = [] + successful_calls = 0 + technical_failures = 0 + started = time.perf_counter() + for case in cases: + case_dir = args.output / case["case_id"].lower() + case_dir.mkdir() + observations = case["observations"] + _write_json(case_dir / "v3_style_input_observations.json", observations) + prompt = build_prompt(case) + (case_dir / "prompt.txt").write_text(prompt, encoding="utf-8") + try: + raw, metadata = call_ollama(args.endpoint, args.model, prompt, args.timeout, args.num_ctx, args.num_predict) + successful_calls += 1 + except Exception as exc: # one recorded attempt; never retry + technical_failures += 1 + failure = {"case_id": case["case_id"], "classification": "FAIL", "technical_failure": True, "error_type": type(exc).__name__, "error": str(exc)} + _write_json(case_dir / "ollama_metadata.json", {"model": args.model, "configuration": {"temperature": 0, "think": False, "num_ctx": args.num_ctx, "num_predict": args.num_predict, "retries": 0}, "technical_failure": failure}) + _write_json(case_dir / "structural_validation.json", {"valid": False, "error": str(exc)}) + _write_json(case_dir / "evaluation.json", failure) + evaluations.append(failure) + continue + (case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8") + _write_json(case_dir / "ollama_metadata.json", metadata) + try: + parsed = parse_model_json(raw) + _write_json(case_dir / "parsed_semantic_classification.json", parsed) + evaluation = evaluate_case(case, parsed) + validation = {"valid": True, "error": None} + except (DerivationValidationError, json.JSONDecodeError) as exc: + validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)} + evaluation = {"case_id": case["case_id"], "classification": "FAIL", "error": str(exc), "normative_leakage": "forbidden" in str(exc)} + _write_json(case_dir / "structural_validation.json", validation) + _write_json(case_dir / "evaluation.json", evaluation) + evaluations.append(evaluation) + summary = { + "experiment": "negative_act_form_v0", "model": args.model, + "successful_llm_call_count": successful_calls, + "technical_failed_call_count": technical_failures, + "runtime_seconds": round(time.perf_counter() - started, 3), + "counts": {label: sum(item["classification"] == label for item in evaluations) for label in ("PASS", "PARTIAL", "FAIL")}, + "evaluations": evaluations, + } + _write_json(args.output / "summary.json", summary) + return summary + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description="Run isolated Negative Act Form experiment") + parser.add_argument("cases", type=Path) + parser.add_argument("-o", "--output", type=Path, required=True) + parser.add_argument("--model", default=DEFAULT_MODEL) + parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT) + parser.add_argument("--timeout", type=int, default=300) + parser.add_argument("--num-ctx", type=int, default=16384) + parser.add_argument("--num-predict", type=int, default=1024) + return parser.parse_args() + + +def main() -> int: + summary = run_experiment(parse_args()) + print(json.dumps(summary, ensure_ascii=False, indent=2)) + return 0 if summary["counts"]["FAIL"] == 0 and summary["technical_failed_call_count"] == 0 else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/gold/negative_act_form_v0/cases.json b/tests/gold/negative_act_form_v0/cases.json new file mode 100644 index 0000000..7b3132c --- /dev/null +++ b/tests/gold/negative_act_form_v0/cases.json @@ -0,0 +1,66 @@ +{ + "schema_version": "experimental-negative-act-form-gold-v0", + "cases": [ + { + "case_id": "NA-01", "description": "Explicit non-pursuit", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Mit Dr. Schlummer arbeiten wir nicht weiter.", "speaker": "Martin", "named_person": "Dr. Schlummer", "addressee": null} + ], + "expected": {"observation_id": "obs_1", "negative_act_form": "explicit_non_pursuit", "action_concepts": [["schlummer"], ["arbeit", "collabor"], ["weiter", "fortsetz", "continu"]]} + }, + { + "case_id": "NA-02", "description": "Explicit non-pursuit paraphrase", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Die externe Lösung verfolgen wir nicht weiter.", "speaker": "Martin", "named_person": null, "addressee": null} + ], + "expected": {"observation_id": "obs_1", "negative_act_form": "explicit_non_pursuit", "action_concepts": [["extern"], ["lösung", "solution"], ["weiter", "pursu", "continu"]]} + }, + { + "case_id": "NA-03", "description": "Personal preference with local context", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Wir könnten die reale Anlage für den Versuch nutzen.", "speaker": "Martin", "named_person": null, "addressee": null}, + {"observation_id": "obs_2", "evidence_id": "e2", "content": "Martin: Ich würde das nicht machen.", "speaker": "Martin", "named_person": null, "addressee": null} + ], + "expected": {"observation_id": "obs_2", "negative_act_form": "personal_preference", "action_concepts": [["real"], ["anlage", "plant"], ["versuch", "trial", "test"]]} + }, + { + "case_id": "NA-04", "description": "Negative recommendation", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Wir könnten die reale Anlage verwenden.", "speaker": "Martin", "named_person": null, "addressee": null}, + {"observation_id": "obs_2", "evidence_id": "e2", "content": "Martin: Ich würde eher davon abraten.", "speaker": "Martin", "named_person": null, "addressee": null} + ], + "expected": {"observation_id": "obs_2", "negative_act_form": "recommendation", "action_concepts": [["real"], ["anlage", "plant"], ["verwend", "use"]]} + }, + { + "case_id": "NA-05", "description": "Temporary non-action", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Wir könnten die Waschstufe einbauen.", "speaker": "Martin", "named_person": null, "addressee": null}, + {"observation_id": "obs_2", "evidence_id": "e2", "content": "Martin: Das machen wir erstmal noch nicht.", "speaker": "Martin", "named_person": null, "addressee": null} + ], + "expected": {"observation_id": "obs_2", "negative_act_form": "temporary_non_action", "action_concepts": [["waschstufe", "washing stage"], ["einbau", "install"]]} + }, + { + "case_id": "NA-06", "description": "Concern only", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Wir könnten das neue Material einsetzen.", "speaker": "Martin", "named_person": null, "addressee": null}, + {"observation_id": "obs_2", "evidence_id": "e2", "content": "Martin: Das wäre kritisch.", "speaker": "Martin", "named_person": null, "addressee": null} + ], + "expected": {"observation_id": "obs_2", "negative_act_form": "none", "action_concepts": []} + }, + { + "case_id": "NA-07", "description": "Uncertainty", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Eine Möglichkeit wäre, die Waschstufe einzubauen.", "speaker": "Martin", "named_person": null, "addressee": null}, + {"observation_id": "obs_2", "evidence_id": "e2", "content": "Martin: Ich weiß nicht, ob das sinnvoll ist.", "speaker": "Martin", "named_person": null, "addressee": null} + ], + "expected": {"observation_id": "obs_2", "negative_act_form": "none", "action_concepts": []} + }, + { + "case_id": "NA-08", "description": "Factual negation", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Das Material ist nicht verfügbar.", "speaker": "Martin", "named_person": null, "addressee": null} + ], + "expected": {"observation_id": "obs_1", "negative_act_form": "none", "action_concepts": []} + } + ] +} diff --git a/tests/test_negative_act_form_experiment.py b/tests/test_negative_act_form_experiment.py new file mode 100644 index 0000000..1f41461 --- /dev/null +++ b/tests/test_negative_act_form_experiment.py @@ -0,0 +1,162 @@ +import argparse +import json +import tempfile +import unittest +from copy import deepcopy +from pathlib import Path +from unittest.mock import patch + +import src.meeting_lab.controlled_semantic_derivation.experiment_negative_act as module +from src.meeting_lab.controlled_semantic_derivation.experiment_negative_act import ( + DerivationValidationError, + build_ollama_payload, + build_prompt, + evaluate_case, + load_gold_cases, + parse_model_json, + run_experiment, + validate_classification, +) + + +GOLD_PATH = Path("tests/gold/negative_act_form_v0/cases.json") + + +FORM_TEXT = { + "NA-01": "Zusammenarbeit mit Dr. Schlummer fortsetzen", + "NA-02": "externe Lösung weiterverfolgen", + "NA-03": "reale Anlage für den Versuch nutzen", + "NA-04": "reale Anlage verwenden", + "NA-05": "Waschstufe einbauen", +} + + +def classification_for(case): + expected = case["expected"] + return { + "observation_id": expected["observation_id"], + "negative_act_form": expected["negative_act_form"], + "normalized_action_text": FORM_TEXT.get(case["case_id"]), + } + + +class NegativeActFormExperimentTests(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.cases = load_gold_cases(GOLD_PATH) + cls.by_id = {case["case_id"]: case for case in cls.cases} + + def test_fixture_contains_exactly_na_01_through_na_08(self): + self.assertEqual(list(self.by_id), [f"NA-{number:02d}" for number in range(1, 9)]) + + def test_exact_schema_is_accepted(self): + case = self.by_id["NA-01"] + self.assertEqual(validate_classification(classification_for(case), case["observations"]), classification_for(case)) + + def test_unknown_field_is_rejected(self): + case = self.by_id["NA-01"] + classification = classification_for(case) + classification["explanation"] = "extra" + with self.assertRaisesRegex(DerivationValidationError, "unknown keys"): + validate_classification(classification, case["observations"]) + + def test_invalid_enum_is_rejected(self): + case = self.by_id["NA-01"] + classification = classification_for(case) + classification["negative_act_form"] = "rejection" + with self.assertRaisesRegex(DerivationValidationError, "unsupported value"): + validate_classification(classification, case["observations"]) + + def test_non_none_requires_normalized_action_text(self): + case = self.by_id["NA-01"] + for value in (None, ""): + classification = classification_for(case) + classification["normalized_action_text"] = value + with self.subTest(value=value), self.assertRaises(DerivationValidationError): + validate_classification(classification, case["observations"]) + + def test_none_requires_null_normalized_action_text(self): + case = self.by_id["NA-06"] + classification = classification_for(case) + self.assertIsNone(classification["normalized_action_text"]) + classification["normalized_action_text"] = "Material einsetzen" + with self.assertRaisesRegex(DerivationValidationError, "requires null"): + validate_classification(classification, case["observations"]) + + def test_forbidden_normative_fields_are_rejected_recursively(self): + case = self.by_id["NA-01"] + fields = ( + "rejection_form", "explicitly_rejected", "status", "decision", "outcome", + "topic_status", "responsible_person", "responsibility", "owner", + "requested_actor", "action_item", "protocol_category", "confidence", + "relation", "relations", "graph", "unresolved_issue", + ) + for field in fields: + classification = classification_for(case) + classification["wrapper"] = {field: "forbidden"} + with self.subTest(field=field), self.assertRaisesRegex(DerivationValidationError, "forbidden semantic keys"): + validate_classification(classification, case["observations"]) + + def test_unknown_observation_id_is_rejected(self): + case = self.by_id["NA-01"] + classification = classification_for(case) + classification["observation_id"] = "obs_99" + with self.assertRaisesRegex(DerivationValidationError, "unknown observation"): + validate_classification(classification, case["observations"]) + + def test_malformed_json_is_rejected(self): + with self.assertRaises(json.JSONDecodeError): + parse_model_json("{bad json") + + def test_all_expected_classifications_evaluate_as_pass(self): + for case in self.cases: + evaluation = evaluate_case(case, classification_for(case)) + with self.subTest(case=case["case_id"]): + self.assertEqual(evaluation["classification"], "PASS") + + def test_fixed_prompt_contains_candidate_and_no_gold_expectation(self): + prompt = build_prompt(self.by_id["NA-03"]) + self.assertIn("candidate observation is obs_2", prompt) + self.assertNotIn("expected", prompt) + self.assertNotIn("Who is responsible", prompt) + + def test_fixed_model_configuration(self): + payload = build_ollama_payload("qwen3.5:9B", "prompt", 16384, 1024) + self.assertFalse(payload["think"]) + self.assertFalse(payload["stream"]) + self.assertEqual(payload["options"]["temperature"], 0) + + def test_no_rejection_or_status_derivation_function_exists(self): + public_names = {name for name in dir(module) if not name.startswith("_")} + self.assertNotIn("derive_rejection", public_names) + self.assertFalse(any(name.startswith("derive_") for name in public_names)) + + def test_artifacts_preserve_semantic_classification_only(self): + case = self.by_id["NA-01"] + raw = json.dumps(classification_for(case), ensure_ascii=False) + with tempfile.TemporaryDirectory() as temporary: + output = Path(temporary) / "run" + args = argparse.Namespace( + cases=GOLD_PATH, output=output, model="qwen3.5:9B", + endpoint="http://unused", timeout=1, num_ctx=16384, num_predict=1024, + ) + with patch.object(module, "load_gold_cases", return_value=[deepcopy(case)]), patch.object( + module, "call_ollama", return_value=(raw, {"model": "qwen3.5:9B"}) + ): + summary = run_experiment(args) + self.assertEqual(summary["successful_llm_call_count"], 1) + case_dir = output / "na-01" + for filename in ( + "v3_style_input_observations.json", "prompt.txt", "raw_model_response.txt", + "parsed_semantic_classification.json", "structural_validation.json", + "evaluation.json", "ollama_metadata.json", + ): + self.assertTrue((case_dir / filename).is_file(), filename) + self.assertFalse((case_dir / "final_derived_result.json").exists()) + self.assertFalse((case_dir / "deterministic_gate_results.json").exists()) + parsed = json.loads((case_dir / "parsed_semantic_classification.json").read_text()) + self.assertEqual(set(parsed), {"observation_id", "negative_act_form", "normalized_action_text"}) + + +if __name__ == "__main__": + unittest.main()