From a1fe89de52cf4b4ee5e53a7b7eb75aa0c33943c0 Mon Sep 17 00:00:00 2001 From: Martin Date: Thu, 20 Aug 2026 09:13:30 +0200 Subject: [PATCH] Add request-acceptance gold experiment --- docs/experiments.md | 67 ++++ .../run_request_acceptance_gold_experiment.py | 16 + .../experiment_gold.py | 367 ++++++++++++++++++ tests/gold/request_acceptance_v0/cases.json | 101 +++++ ...test_request_acceptance_gold_experiment.py | 178 +++++++++ 5 files changed, 729 insertions(+) create mode 100644 scripts/run_request_acceptance_gold_experiment.py create mode 100644 src/meeting_lab/controlled_semantic_derivation/experiment_gold.py create mode 100644 tests/gold/request_acceptance_v0/cases.json create mode 100644 tests/test_request_acceptance_gold_experiment.py diff --git a/docs/experiments.md b/docs/experiments.md index c45379b..3a40611 100644 --- a/docs/experiments.md +++ b/docs/experiments.md @@ -1680,6 +1680,73 @@ does not generalize the derivation architecture to other cases or semantic categories. No production integration, other case run, semantic graph, protocol derivation or Progeo run occurred. +## EXP-0032 — Request / Acceptance Gold V0 + +Status: Experimental; promising with semantic precision gaps + +Date: 2026-08-20 + +This isolated regression experiment tested whether the EXP-0031 mechanism +generalizes beyond H. It used ten short synthetic cases containing only +V3-style observations. Evidence Observation V3 was neither called nor changed, +and the model received no raw transcript or expected result. The fixed +recognition schema permits only a nullable concrete request and nullable later +explicit personal commitment, plus the same-requested-work judgment and +normalized action text. Responsibility, requested actor, established status, +Action Item, protocol, confidence and generic graph fields remain forbidden. + +Cases: + +- RA-01 explicit positive acceptance: PASS. +- RA-02 paraphrased positive acceptance: PASS. +- RA-03 acknowledgement only: PASS. +- RA-04 tentative response: PASS. +- RA-05 different responder without personal acceptance: PASS. +- RA-06 explicit commitment to different work: PARTIAL. The model returned no + acceptance instead of recognizing a commitment with `same_requested_work` + false. The requested action correctly remained unestablished. +- RA-07 request without response: PASS. +- RA-08 collective commitment: PARTIAL. The model over-recognized the + collective `wir` statement as an explicit commitment, but no request existed + and deterministic gates prevented individual responsibility. +- RA-09 impersonal necessity: PARTIAL. The model over-recognized the impersonal + necessity as a concrete request, but the observation had no addressee and + deterministic gates prevented establishment. +- RA-10 tentative personal suggestion: PASS. + +Configuration: exactly ten sequential `qwen3.5:9B` calls, one per case, +temperature 0, `think=false`, `num_ctx=16384`, `num_predict=1024`, no retries, +no voting and no prompt change between cases. Summed call time was 23.754 +seconds, with 4,826 prompt-evaluation tokens and 793 evaluation tokens. The +strict schema validated every response and no responsibility or establishment +field leaked into model output. + +Both positive cases recognized the request, explicit commitment and same-work +relationship, including the paraphrased acceptance, and deterministically +established Clara as responsible with due date `Dienstag`. The model rendered +the normalized action in semantically equivalent English; evaluation therefore +checks the structural deterministic result exactly while treating normalized +action wording as evidence-near semantic text rather than requiring lexical +identity. Acknowledgement and tentative response were not promoted. Every +negative case remained unestablished, and no individual responsibility was +invented. + +Recognition-level errors were two false positives (RA-08 commitment and RA-09 +request) and one false negative (RA-06 different-work commitment). Final +established-action false positives and false negatives were both zero. The +overall result was seven PASS, three PARTIAL and zero FAIL. + +Conclusion: the narrow request-plus-acceptance architecture remains promising +for established individual actions because deterministic addressee, ordering, +speaker, same-work, provenance and deadline gates contained all recognition +errors. The recognition layer is not yet precise enough to generalize: its +handling of collective commitment, impersonal necessity and commitments to +different work needs further isolated study. No production integration or +additional semantic category is justified by this result. + +Artifacts are preserved under +`artifacts/experiments/request_acceptance_gold_v0/20260820_qwen35_9b_single_run/`. + ## EXP-0026 — Topic-oriented Discussion Subject reconstruction V2 prototype Date: 2026-08-11 diff --git a/scripts/run_request_acceptance_gold_experiment.py b/scripts/run_request_acceptance_gold_experiment.py new file mode 100644 index 0000000..5eff9c0 --- /dev/null +++ b/scripts/run_request_acceptance_gold_experiment.py @@ -0,0 +1,16 @@ +#!/usr/bin/env python3 +"""Repository entry point for the request/acceptance Gold experiment.""" + +import sys +from pathlib import Path + + +REPO_ROOT = Path(__file__).resolve().parents[1] +if str(REPO_ROOT) not in sys.path: + sys.path.insert(0, str(REPO_ROOT)) + +from src.meeting_lab.controlled_semantic_derivation.experiment_gold import main # noqa: E402 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/meeting_lab/controlled_semantic_derivation/experiment_gold.py b/src/meeting_lab/controlled_semantic_derivation/experiment_gold.py new file mode 100644 index 0000000..bcafd02 --- /dev/null +++ b/src/meeting_lab/controlled_semantic_derivation/experiment_gold.py @@ -0,0 +1,367 @@ +#!/usr/bin/env python3 +"""Isolated request/acceptance Gold reliability experiment.""" + +from __future__ import annotations + +import argparse +import json +import re +import time +from pathlib import Path +from typing import Any + +from .experiment_h import ( + DEFAULT_ENDPOINT, + DEFAULT_MODEL, + DerivationValidationError, + FORBIDDEN_LLM_KEYS, + OBSERVATION_KEYS, + call_ollama, +) + + +GOLD_SCHEMA_VERSION = "experimental-request-acceptance-gold-v0" +RECOGNITION_SCHEMA_VERSION = "experimental-request-acceptance-recognition-v0" +REQUEST_KEYS = {"observation_id", "is_concrete_request", "normalized_action_text"} +ACCEPTANCE_KEYS = { + "observation_id", "is_explicit_commitment", "same_requested_work", + "normalized_action_text", +} +WEEKDAYS = { + "monday": "Montag", "montag": "Montag", "tuesday": "Dienstag", + "dienstag": "Dienstag", "wednesday": "Mittwoch", "mittwoch": "Mittwoch", + "thursday": "Donnerstag", "donnerstag": "Donnerstag", "friday": "Freitag", + "freitag": "Freitag", "saturday": "Samstag", "samstag": "Samstag", + "sunday": "Sonntag", "sonntag": "Sonntag", +} + +PROMPT_TEMPLATE = """Recognize only a concrete directed request and a later explicit personal commitment in the supplied V3-style observations. + +The input is observations only, not a transcript. Identify: +1. A concrete request directed to the observation's explicit addressee, if one exists. +2. A later response that explicitly commits its speaker to work, if one exists. +3. Whether that explicit commitment concerns substantially the same requested work. + +Lexical identity is not required: a contextual paraphrase may denote the same work. Mere acknowledgement, tentative or conditional language, collective "we" statements, impersonal necessity, suggestions, and statements that work should be done are not explicit personal commitments. A commitment to different work is an explicit commitment but not the same requested work. + +Do not decide or output responsibility, requested actor, established status, Action Item status, protocol eligibility, confidence, semantic relations, or graphs. Do not answer who is responsible. Deterministic code applies those gates later. + +Return exactly this JSON shape and no other fields. Use null for request or acceptance when no qualifying observation exists: +{{ + "schema_version": "experimental-request-acceptance-recognition-v0", + "request": null | {{ + "observation_id": "observation ID", + "is_concrete_request": true, + "normalized_action_text": "concise requested work" + }}, + "acceptance": null | {{ + "observation_id": "observation ID", + "is_explicit_commitment": true, + "same_requested_work": true, + "normalized_action_text": "concise committed work" + }} +}} + +V3-style observations: +{observations_json} +""" + + +def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None: + missing = required - value.keys() + unknown = value.keys() - required + if missing: + raise DerivationValidationError(f"{location} missing required keys: {sorted(missing)}") + if unknown: + raise DerivationValidationError(f"{location} has unknown keys: {sorted(unknown)}") + + +def _nonempty_text(value: Any, location: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise DerivationValidationError(f"{location} must be a non-empty string") + return value.strip() + + +def load_gold_cases(path: Path) -> list[dict[str, Any]]: + data = json.loads(path.read_text(encoding="utf-8-sig")) + if not isinstance(data, dict): + raise DerivationValidationError("Gold fixture must be an object") + _exact_keys(data, {"schema_version", "cases"}, "Gold fixture") + if data["schema_version"] != GOLD_SCHEMA_VERSION: + raise DerivationValidationError("unexpected Gold fixture schema_version") + cases = data["cases"] + if not isinstance(cases, list) or not cases: + raise DerivationValidationError("Gold fixture cases must be a non-empty list") + seen_cases: set[str] = set() + for case in cases: + _exact_keys(case, {"case_id", "description", "observations", "expected_recognition", "expected_result"}, "Gold case") + case_id = _nonempty_text(case["case_id"], "case_id") + if case_id in seen_cases: + raise DerivationValidationError(f"duplicate case ID: {case_id}") + seen_cases.add(case_id) + _validate_observations(case["observations"]) + return cases + + +def _validate_observations(observations: Any) -> None: + if not isinstance(observations, list) or not observations: + raise DerivationValidationError("observations must be a non-empty list") + seen_ids: set[str] = set() + seen_evidence: set[str] = set() + for index, observation in enumerate(observations): + location = f"observations[{index}]" + if not isinstance(observation, dict): + raise DerivationValidationError(f"{location} must be an object") + _exact_keys(observation, OBSERVATION_KEYS, location) + observation_id = _nonempty_text(observation["observation_id"], f"{location}.observation_id") + evidence_id = _nonempty_text(observation["evidence_id"], f"{location}.evidence_id") + if observation_id in seen_ids or evidence_id in seen_evidence: + raise DerivationValidationError("observation and evidence IDs must be unique") + seen_ids.add(observation_id) + seen_evidence.add(evidence_id) + _nonempty_text(observation["content"], f"{location}.content") + _nonempty_text(observation["speaker"], f"{location}.speaker") + for field in ("named_person", "addressee"): + if observation[field] is not None: + _nonempty_text(observation[field], f"{location}.{field}") + + +def build_prompt(observations: list[dict[str, Any]]) -> str: + _validate_observations(observations) + return PROMPT_TEMPLATE.format( + observations_json=json.dumps(observations, ensure_ascii=False, indent=2) + ) + + +def parse_model_json(raw_text: str) -> dict[str, Any]: + data = json.loads(raw_text) + if not isinstance(data, dict): + raise DerivationValidationError("semantic recognition must be an object") + return data + + +def _reject_forbidden_keys(value: Any, location: str = "output") -> None: + if isinstance(value, dict): + forbidden = FORBIDDEN_LLM_KEYS.intersection(value) + if forbidden: + raise DerivationValidationError( + f"{location} contains forbidden semantic keys: {sorted(forbidden)}" + ) + for key, item in value.items(): + _reject_forbidden_keys(item, f"{location}.{key}") + elif isinstance(value, list): + for index, item in enumerate(value): + _reject_forbidden_keys(item, f"{location}[{index}]") + + +def validate_recognition(data: Any, observations: list[dict[str, Any]]) -> dict[str, Any]: + if not isinstance(data, dict): + raise DerivationValidationError("semantic recognition must be an object") + _reject_forbidden_keys(data) + _exact_keys(data, {"schema_version", "request", "acceptance"}, "output") + if data["schema_version"] != RECOGNITION_SCHEMA_VERSION: + raise DerivationValidationError("unexpected recognition schema_version") + known_ids = {item["observation_id"] for item in observations} + request = data["request"] + acceptance = data["acceptance"] + if request is not None: + if not isinstance(request, dict): + raise DerivationValidationError("output.request must be an object or null") + _exact_keys(request, REQUEST_KEYS, "output.request") + if request["observation_id"] not in known_ids: + raise DerivationValidationError("request references unknown observation") + if not isinstance(request["is_concrete_request"], bool): + raise DerivationValidationError("is_concrete_request must be boolean") + _nonempty_text(request["normalized_action_text"], "request.normalized_action_text") + if acceptance is not None: + if not isinstance(acceptance, dict): + raise DerivationValidationError("output.acceptance must be an object or null") + _exact_keys(acceptance, ACCEPTANCE_KEYS, "output.acceptance") + if acceptance["observation_id"] not in known_ids: + raise DerivationValidationError("acceptance references unknown observation") + for field in ("is_explicit_commitment", "same_requested_work"): + if not isinstance(acceptance[field], bool): + raise DerivationValidationError(f"{field} must be boolean") + _nonempty_text(acceptance["normalized_action_text"], "acceptance.normalized_action_text") + if request is not None and acceptance is not None and request["observation_id"] == acceptance["observation_id"]: + raise DerivationValidationError("request and acceptance must reference different observations") + return data + + +def _bounded_due(observations: list[dict[str, Any]]) -> tuple[str | None, bool]: + forms: set[str] = set() + for observation in observations: + for token in re.findall(r"\b[A-Za-zÄÖÜäöü]+\b", observation["content"].casefold()): + if token in WEEKDAYS: + forms.add(WEEKDAYS[token]) + return (next(iter(forms)) if len(forms) == 1 else None, len(forms) <= 1) + + +def _strip_due(action_text: str) -> str: + weekday = "|".join(re.escape(value) for value in WEEKDAYS) + result = re.sub(rf"\s+(?:bis|by)\s+(?:{weekday})\b", "", action_text, flags=re.IGNORECASE) + return result.strip(" .,:;-") or action_text.strip() + + +def derive_action( + observations: list[dict[str, Any]], recognition: dict[str, Any] +) -> tuple[dict[str, bool], dict[str, Any] | None]: + _validate_observations(observations) + validate_recognition(recognition, observations) + by_id = {item["observation_id"]: item for item in observations} + positions = {item["observation_id"]: index for index, item in enumerate(observations)} + request_semantic = recognition["request"] + acceptance_semantic = recognition["acceptance"] + request = by_id.get(request_semantic["observation_id"]) if request_semantic else None + acceptance = by_id.get(acceptance_semantic["observation_id"]) if acceptance_semantic else None + due, deadline_consistent = _bounded_due(observations) + gates = { + "request_semantic_positive": request_semantic is not None and request_semantic["is_concrete_request"] is True, + "request_observation_exists": request is not None, + "request_has_addressee": request is not None and isinstance(request["addressee"], str) and bool(request["addressee"].strip()), + "acceptance_semantic_positive": acceptance_semantic is not None and acceptance_semantic["is_explicit_commitment"] is True, + "same_requested_work": acceptance_semantic is not None and acceptance_semantic["same_requested_work"] is True, + "acceptance_observation_exists": acceptance is not None, + "acceptance_after_request": request is not None and acceptance is not None and positions[acceptance["observation_id"]] > positions[request["observation_id"]], + "acceptance_speaker_matches_addressee": request is not None and acceptance is not None and acceptance["speaker"] == request["addressee"], + "provenance_valid_and_consistent": request is not None and acceptance is not None and request["evidence_id"] in {item["evidence_id"] for item in observations} and acceptance["evidence_id"] in {item["evidence_id"] for item in observations}, + "deadline_consistent": deadline_consistent, + } + if not all(gates.values()): + return gates, None + return gates, { + "action_id": "action_1", + "content": _strip_due(request_semantic["normalized_action_text"]), + "status": "established", + "requested_actor": request["addressee"], + "responsible_person": acceptance["speaker"], + "due": due, + "support": { + "request": {"observation_id": request["observation_id"], "evidence_id": request["evidence_id"]}, + "acceptance": {"observation_id": acceptance["observation_id"], "evidence_id": acceptance["evidence_id"]}, + }, + } + + +def evaluate_case(case: dict[str, Any], recognition: dict[str, Any]) -> dict[str, Any]: + observations = case["observations"] + validate_recognition(recognition, observations) + gates, result = derive_action(observations, recognition) + expected_recognition = case["expected_recognition"] + request_correct = (recognition["request"] is not None and recognition["request"]["is_concrete_request"]) == expected_recognition["request"] + commitment_correct = (recognition["acceptance"] is not None and recognition["acceptance"]["is_explicit_commitment"]) == expected_recognition["commitment"] + same_work_correct = (recognition["acceptance"] is not None and recognition["acceptance"]["same_requested_work"]) == expected_recognition["same_work"] + expected = case["expected_result"] + actual_established = result is not None + final_correct = actual_established == expected["established"] + if result is not None: + final_correct = final_correct and all( + result[key] == expected[key] + for key in ("requested_actor", "responsible_person", "due") + ) + else: + final_correct = final_correct and expected["responsible_person"] is None + semantic_correct = request_correct and commitment_correct and same_work_correct + classification = "PASS" if final_correct and semantic_correct else ("PARTIAL" if final_correct else "FAIL") + return { + "case_id": case["case_id"], "classification": classification, + "request_correct": request_correct, "commitment_correct": commitment_correct, + "same_work_correct": same_work_correct, "deterministic_gates_correct": final_correct, + "final_result_correct": final_correct, "result": result, "gates": gates, + "unsupported_semantic_strengthening": not semantic_correct and actual_established, + "responsibility_status_leakage": False, + } + + +def reevaluate_existing(args: argparse.Namespace) -> dict[str, Any]: + cases = load_gold_cases(args.cases) + evaluations = [] + for case in cases: + case_dir = args.output / case["case_id"].lower() + parsed = json.loads( + (case_dir / "parsed_semantic_recognition.json").read_text(encoding="utf-8") + ) + evaluation = evaluate_case(case, parsed) + _write_json(case_dir / "evaluation.json", evaluation) + evaluations.append(evaluation) + metadata = [ + json.loads( + (args.output / case["case_id"].lower() / "ollama_metadata.json").read_text( + encoding="utf-8" + ) + ) + for case in cases + ] + summary = { + "experiment": "request_acceptance_gold_v0", "model": args.model, + "llm_call_count": len(cases), + "runtime_seconds": round(sum(item["elapsed_seconds"] for item in metadata), 3), + "counts": {label: sum(item["classification"] == label for item in evaluations) for label in ("PASS", "PARTIAL", "FAIL")}, + "evaluations": evaluations, + } + _write_json(args.output / "summary.json", summary) + return summary + + +def _write_json(path: Path, value: Any) -> None: + path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + + +def run_gold(args: argparse.Namespace) -> dict[str, Any]: + cases = load_gold_cases(args.cases) + args.output.mkdir(parents=True, exist_ok=False) + _write_json(args.output / "gold_cases.json", {"schema_version": GOLD_SCHEMA_VERSION, "cases": cases}) + evaluations = [] + total_started = time.perf_counter() + for case in cases: + case_dir = args.output / case["case_id"].lower() + case_dir.mkdir() + _write_json(case_dir / "v3_style_input_observations.json", case["observations"]) + prompt = build_prompt(case["observations"]) + (case_dir / "prompt.txt").write_text(prompt, encoding="utf-8") + raw, metadata = call_ollama(args.endpoint, args.model, prompt, args.timeout, args.num_ctx, args.num_predict) + (case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8") + _write_json(case_dir / "ollama_metadata.json", metadata) + parsed = parse_model_json(raw) + _write_json(case_dir / "parsed_semantic_recognition.json", parsed) + try: + evaluation = evaluate_case(case, parsed) + validation = {"valid": True, "error": None} + except (DerivationValidationError, json.JSONDecodeError) as exc: + validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)} + evaluation = {"case_id": case["case_id"], "classification": "FAIL", "error": str(exc), "responsibility_status_leakage": "forbidden" in str(exc)} + _write_json(case_dir / "structural_validation.json", validation) + _write_json(case_dir / "evaluation.json", evaluation) + evaluations.append(evaluation) + summary = { + "experiment": "request_acceptance_gold_v0", "model": args.model, + "llm_call_count": len(cases), "runtime_seconds": round(time.perf_counter() - total_started, 3), + "counts": {label: sum(item["classification"] == label for item in evaluations) for label in ("PASS", "PARTIAL", "FAIL")}, + "evaluations": evaluations, + } + _write_json(args.output / "summary.json", summary) + return summary + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description="Run isolated request/acceptance Gold experiment") + parser.add_argument("cases", type=Path) + parser.add_argument("-o", "--output", type=Path, required=True) + parser.add_argument("--model", default=DEFAULT_MODEL) + parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT) + parser.add_argument("--timeout", type=int, default=300) + parser.add_argument("--num-ctx", type=int, default=16384) + parser.add_argument("--num-predict", type=int, default=1024) + parser.add_argument("--reevaluate-existing", action="store_true") + return parser.parse_args() + + +def main() -> int: + args = parse_args() + summary = reevaluate_existing(args) if args.reevaluate_existing else run_gold(args) + print(json.dumps(summary, ensure_ascii=False, indent=2)) + return 0 if summary["counts"]["FAIL"] == 0 else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/gold/request_acceptance_v0/cases.json b/tests/gold/request_acceptance_v0/cases.json new file mode 100644 index 0000000..9b6b630 --- /dev/null +++ b/tests/gold/request_acceptance_v0/cases.json @@ -0,0 +1,101 @@ +{ + "schema_version": "experimental-request-acceptance-gold-v0", + "cases": [ + { + "case_id": "RA-01", + "description": "Explicit positive acceptance", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Antonius: Clara, kannst du die Messwerte bis Dienstag auswerten?", "speaker": "Antonius", "named_person": "Clara", "addressee": "Clara"}, + {"observation_id": "obs_2", "evidence_id": "e2", "content": "Clara: Ja, ich übernehme die Auswertung bis Dienstag.", "speaker": "Clara", "named_person": null, "addressee": null} + ], + "expected_recognition": {"request": true, "commitment": true, "same_work": true}, + "expected_result": {"established": true, "content": "Auswertung der Messwerte", "requested_actor": "Clara", "responsible_person": "Clara", "due": "Dienstag"} + }, + { + "case_id": "RA-02", + "description": "Paraphrased positive acceptance", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Antonius: Clara, kannst du die Messwerte bis Dienstag auswerten?", "speaker": "Antonius", "named_person": "Clara", "addressee": "Clara"}, + {"observation_id": "obs_2", "evidence_id": "e2", "content": "Clara: Ja. Ich kümmere mich darum und habe die Auswertung bis Dienstag fertig.", "speaker": "Clara", "named_person": null, "addressee": null} + ], + "expected_recognition": {"request": true, "commitment": true, "same_work": true}, + "expected_result": {"established": true, "content": "Auswertung der Messwerte", "requested_actor": "Clara", "responsible_person": "Clara", "due": "Dienstag"} + }, + { + "case_id": "RA-03", + "description": "Acknowledgement only", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Antonius: Clara, kannst du die Messwerte bis Dienstag auswerten?", "speaker": "Antonius", "named_person": "Clara", "addressee": "Clara"}, + {"observation_id": "obs_2", "evidence_id": "e2", "content": "Clara: Ja, ich habe verstanden, worum es geht.", "speaker": "Clara", "named_person": null, "addressee": null} + ], + "expected_recognition": {"request": true, "commitment": false, "same_work": false}, + "expected_result": {"established": false, "content": null, "requested_actor": null, "responsible_person": null, "due": null} + }, + { + "case_id": "RA-04", + "description": "Tentative response", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Antonius: Clara, kannst du die Messwerte bis Dienstag auswerten?", "speaker": "Antonius", "named_person": "Clara", "addressee": "Clara"}, + {"observation_id": "obs_2", "evidence_id": "e2", "content": "Clara: Ich schaue mal, ob ich das schaffe.", "speaker": "Clara", "named_person": null, "addressee": null} + ], + "expected_recognition": {"request": true, "commitment": false, "same_work": false}, + "expected_result": {"established": false, "content": null, "requested_actor": null, "responsible_person": null, "due": null} + }, + { + "case_id": "RA-05", + "description": "Different responder without personal acceptance", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Antonius: Clara, kannst du die Messwerte bis Dienstag auswerten?", "speaker": "Antonius", "named_person": "Clara", "addressee": "Clara"}, + {"observation_id": "obs_2", "evidence_id": "e2", "content": "Martin: Ja, das sollte gemacht werden.", "speaker": "Martin", "named_person": null, "addressee": null} + ], + "expected_recognition": {"request": true, "commitment": false, "same_work": false}, + "expected_result": {"established": false, "content": null, "requested_actor": null, "responsible_person": null, "due": null} + }, + { + "case_id": "RA-06", + "description": "Explicit commitment to different work", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Antonius: Clara, kannst du die Messwerte bis Dienstag auswerten?", "speaker": "Antonius", "named_person": "Clara", "addressee": "Clara"}, + {"observation_id": "obs_2", "evidence_id": "e2", "content": "Clara: Ja, ich kümmere mich um die Präsentation.", "speaker": "Clara", "named_person": null, "addressee": null} + ], + "expected_recognition": {"request": true, "commitment": true, "same_work": false}, + "expected_result": {"established": false, "content": null, "requested_actor": null, "responsible_person": null, "due": null} + }, + { + "case_id": "RA-07", + "description": "Request without response", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Antonius: Clara, kannst du die Messwerte bis Dienstag auswerten?", "speaker": "Antonius", "named_person": "Clara", "addressee": "Clara"} + ], + "expected_recognition": {"request": true, "commitment": false, "same_work": false}, + "expected_result": {"established": false, "content": null, "requested_actor": null, "responsible_person": null, "due": null} + }, + { + "case_id": "RA-08", + "description": "Collective commitment", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Ja, wir testen nächste Woche 20 Meter.", "speaker": "Martin", "named_person": null, "addressee": null} + ], + "expected_recognition": {"request": false, "commitment": false, "same_work": false}, + "expected_result": {"established": false, "content": null, "requested_actor": null, "responsible_person": null, "due": null} + }, + { + "case_id": "RA-09", + "description": "Impersonal necessity", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Man müsste zuerst prüfen, welches Reinigungsverfahren verfügbar ist.", "speaker": "Martin", "named_person": null, "addressee": null} + ], + "expected_recognition": {"request": false, "commitment": false, "same_work": false}, + "expected_result": {"established": false, "content": null, "requested_actor": null, "responsible_person": null, "due": null} + }, + { + "case_id": "RA-10", + "description": "Personal suggestion", + "observations": [ + {"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Ich würde vielleicht Dirk Textor kontaktieren.", "speaker": "Martin", "named_person": "Dirk Textor", "addressee": null} + ], + "expected_recognition": {"request": false, "commitment": false, "same_work": false}, + "expected_result": {"established": false, "content": null, "requested_actor": null, "responsible_person": null, "due": null} + } + ] +} diff --git a/tests/test_request_acceptance_gold_experiment.py b/tests/test_request_acceptance_gold_experiment.py new file mode 100644 index 0000000..875fc88 --- /dev/null +++ b/tests/test_request_acceptance_gold_experiment.py @@ -0,0 +1,178 @@ +import json +import tempfile +import unittest +from copy import deepcopy +from pathlib import Path + +from src.meeting_lab.controlled_semantic_derivation.experiment_gold import ( + RECOGNITION_SCHEMA_VERSION, + DerivationValidationError, + build_prompt, + derive_action, + evaluate_case, + load_gold_cases, + validate_recognition, +) + + +GOLD_PATH = Path("tests/gold/request_acceptance_v0/cases.json") + + +def recognition_for(case): + expected = case["expected_recognition"] + request = None + acceptance = None + if expected["request"]: + request = { + "observation_id": "obs_1", + "is_concrete_request": True, + "normalized_action_text": "Auswertung der Messwerte bis Dienstag", + } + if expected["commitment"]: + acceptance = { + "observation_id": "obs_2", + "is_explicit_commitment": True, + "same_requested_work": expected["same_work"], + "normalized_action_text": ( + "Auswertung der Messwerte" if expected["same_work"] else "Präsentation" + ), + } + return { + "schema_version": RECOGNITION_SCHEMA_VERSION, + "request": request, + "acceptance": acceptance, + } + + +class RequestAcceptanceGoldExperimentTests(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.cases = load_gold_cases(GOLD_PATH) + cls.by_id = {case["case_id"]: case for case in cls.cases} + + def test_fixture_has_exactly_required_ten_cases(self): + self.assertEqual(list(self.by_id), [f"RA-{number:02d}" for number in range(1, 11)]) + + def test_all_cases_use_minimal_v3_style_observations(self): + required = {"observation_id", "evidence_id", "content", "speaker", "named_person", "addressee"} + for case in self.cases: + with self.subTest(case=case["case_id"]): + self.assertGreaterEqual(len(case["observations"]), 1) + self.assertLessEqual(len(case["observations"]), 2) + self.assertTrue(all(set(observation) == required for observation in case["observations"])) + + def test_positive_cases_establish_exact_action_person_due_and_provenance(self): + for case_id in ("RA-01", "RA-02"): + case = self.by_id[case_id] + gates, result = derive_action(case["observations"], recognition_for(case)) + with self.subTest(case=case_id): + self.assertTrue(all(gates.values())) + self.assertEqual(result["content"], "Auswertung der Messwerte") + self.assertEqual(result["requested_actor"], "Clara") + self.assertEqual(result["responsible_person"], "Clara") + self.assertEqual(result["due"], "Dienstag") + self.assertEqual(result["support"]["request"], {"observation_id": "obs_1", "evidence_id": "e1"}) + self.assertEqual(result["support"]["acceptance"], {"observation_id": "obs_2", "evidence_id": "e2"}) + + def test_paraphrase_does_not_require_lexical_identity(self): + case = self.by_id["RA-02"] + recognition = recognition_for(case) + recognition["acceptance"]["normalized_action_text"] = "darum kümmern und fertigstellen" + gates, result = derive_action(case["observations"], recognition) + self.assertTrue(gates["same_requested_work"]) + self.assertIsNotNone(result) + + def test_acknowledgement_is_unestablished(self): + self._assert_unestablished("RA-03", "acceptance_semantic_positive") + + def test_tentative_response_is_unestablished(self): + self._assert_unestablished("RA-04", "acceptance_semantic_positive") + + def test_different_responder_is_not_personal_acceptance(self): + self._assert_unestablished("RA-05", "acceptance_semantic_positive") + case = self.by_id["RA-05"] + recognition = recognition_for(self.by_id["RA-01"]) + gates, result = derive_action(case["observations"], recognition) + self.assertFalse(gates["acceptance_speaker_matches_addressee"]) + self.assertIsNone(result) + + def test_different_work_fails_same_work_gate(self): + self._assert_unestablished("RA-06", "same_requested_work") + + def test_request_without_response_is_unestablished(self): + self._assert_unestablished("RA-07", "acceptance_observation_exists") + + def test_collective_impersonal_and_suggestion_controls_are_unestablished(self): + for case_id in ("RA-08", "RA-09", "RA-10"): + with self.subTest(case=case_id): + self._assert_unestablished(case_id, "request_semantic_positive") + + def test_named_person_without_request_cannot_create_responsibility(self): + case = self.by_id["RA-10"] + self.assertEqual(case["observations"][0]["named_person"], "Dirk Textor") + _, result = derive_action(case["observations"], recognition_for(case)) + self.assertIsNone(result) + + def test_all_expected_recognitions_evaluate_as_pass(self): + for case in self.cases: + evaluation = evaluate_case(case, recognition_for(case)) + with self.subTest(case=case["case_id"]): + self.assertEqual(evaluation["classification"], "PASS") + + def test_equivalent_translated_normalized_action_does_not_fail_structure(self): + case = self.by_id["RA-01"] + recognition = recognition_for(case) + recognition["request"]["normalized_action_text"] = "evaluate measurement values" + evaluation = evaluate_case(case, recognition) + self.assertEqual(evaluation["classification"], "PASS") + self.assertEqual(evaluation["result"]["content"], "evaluate measurement values") + + def test_forbidden_llm_fields_are_rejected_recursively(self): + case = self.by_id["RA-01"] + for field in ( + "responsible_person", "responsibility", "requested_actor", "status", + "established", "action_item", "protocol_category", "confidence", "graph", + ): + recognition = recognition_for(case) + recognition["request"][field] = "forbidden" + with self.subTest(field=field), self.assertRaisesRegex(DerivationValidationError, "forbidden semantic keys"): + validate_recognition(recognition, case["observations"]) + + def test_unknown_observation_reference_is_rejected(self): + case = self.by_id["RA-01"] + recognition = recognition_for(case) + recognition["acceptance"]["observation_id"] = "obs_99" + with self.assertRaisesRegex(DerivationValidationError, "unknown observation"): + validate_recognition(recognition, case["observations"]) + + def test_duplicate_evidence_provenance_is_rejected(self): + fixture = json.loads(GOLD_PATH.read_text()) + fixture["cases"][0]["observations"][1]["evidence_id"] = "e1" + with tempfile.TemporaryDirectory() as temporary: + path = Path(temporary) / "cases.json" + path.write_text(json.dumps(fixture), encoding="utf-8") + with self.assertRaisesRegex(DerivationValidationError, "must be unique"): + load_gold_cases(path) + + def test_prompt_is_fixed_narrow_and_contains_observations_only(self): + prompt = build_prompt(self.by_id["RA-01"]["observations"]) + self.assertIn("V3-style observations", prompt) + self.assertNotIn("expected_result", prompt) + self.assertNotIn("Who is responsible", prompt) + + def test_conflicting_weekdays_fail_deadline_gate(self): + case = deepcopy(self.by_id["RA-01"]) + case["observations"][1]["content"] = "Clara: Ja, ich übernehme die Auswertung bis Mittwoch." + gates, result = derive_action(case["observations"], recognition_for(case)) + self.assertFalse(gates["deadline_consistent"]) + self.assertIsNone(result) + + def _assert_unestablished(self, case_id, failed_gate): + case = self.by_id[case_id] + gates, result = derive_action(case["observations"], recognition_for(case)) + self.assertFalse(gates[failed_gate]) + self.assertIsNone(result) + + +if __name__ == "__main__": + unittest.main()