Add collective commitment gold experiment
This commit is contained in:
@@ -1747,6 +1747,70 @@ additional semantic category is justified by this result.
|
||||
Artifacts are preserved under
|
||||
`artifacts/experiments/request_acceptance_gold_v0/20260820_qwen35_9b_single_run/`.
|
||||
|
||||
## EXP-0033 — Collective Commitment Gold V0
|
||||
|
||||
Status: Experimental; architecturally successful with one contained
|
||||
recognition false positive
|
||||
|
||||
Date: 2026-08-20
|
||||
|
||||
This isolated second-stage experiment tested whether an explicit collective
|
||||
first-person commitment can establish an action without inventing an individual
|
||||
owner. It used ten synthetic cases containing one minimal V3-style observation
|
||||
each. Evidence Observation V3 was neither called nor changed, and the accepted
|
||||
Request/Acceptance mechanism remained unchanged and independent.
|
||||
|
||||
The strict semantic schema contains exactly `observation_id`,
|
||||
`commitment_form` and `normalized_action_text`. `commitment_form` is closed to
|
||||
`individual_first_person`, `collective_first_person` and `none`. The model
|
||||
cannot output responsibility, ownership, requested actor, establishment,
|
||||
Action Item, protocol, confidence, relations, graphs, decisions or unresolved
|
||||
issues. Deterministic code validates schema and provenance, requires collective
|
||||
commitment plus non-empty action text, applies bounded deadline consistency and
|
||||
explicit-negation gates, and only then sets `status: established`,
|
||||
`commitment_scope: collective` and `responsible_person: null`.
|
||||
|
||||
Gold results:
|
||||
|
||||
- CC-01 explicit collective commitment: PASS; established, due `nächste
|
||||
Woche`, no person.
|
||||
- CC-02 individual commitment: PASS; correctly routed out of the collective
|
||||
path.
|
||||
- CC-03 tentative collective possibility: PASS; unestablished.
|
||||
- CC-04 collective suggestion: PASS; unestablished.
|
||||
- CC-05 impersonal necessity: PASS; unestablished.
|
||||
- CC-06 passive future statement: PASS; unestablished.
|
||||
- CC-07 collective rejection: PARTIAL. The model incorrectly returned
|
||||
`collective_first_person`, but the deterministic negation gate detected
|
||||
`nicht` and prevented establishment.
|
||||
- CC-08 qualified collective commitment: PASS; established with `nur im
|
||||
Technikum` preserved, null due and no person.
|
||||
- CC-09 collective commitment without deadline: PASS; established with null
|
||||
due and no person.
|
||||
- CC-10 speaker ownership trap: PASS; established collectively while Martin
|
||||
remained only the speaker and was not assigned ownership.
|
||||
|
||||
Configuration: exactly ten successful sequential `qwen3.5:9B` calls, one per
|
||||
case, temperature 0, `think=false`, `num_ctx=16384`, `num_predict=1024`, no
|
||||
retries, no voting and no prompt change. There were zero technical failed
|
||||
calls. Aggregate runner time was 10.504 seconds; summed per-call time was 10.500
|
||||
seconds, with 4,267 prompt-evaluation tokens and 415 evaluation tokens.
|
||||
|
||||
The outcome was nine PASS, one PARTIAL and zero FAIL. There was one recognition
|
||||
false positive and no recognition false negatives. No qualifier was lost, no
|
||||
individual owner was invented, and no responsibility or status field leaked
|
||||
into recognition. Bounded due handling preserved `nächste Woche` verbatim and
|
||||
returned null when no deadline was present.
|
||||
|
||||
Conclusion: the collective-commitment path is architecturally successful for
|
||||
this narrow Gold set. The deterministic negation gate contained the only model
|
||||
error, and every successful collective result necessarily retained
|
||||
`responsible_person: null`. This does not justify a generic commitment system,
|
||||
production integration, group identity inference or another semantic category.
|
||||
|
||||
Artifacts are preserved under
|
||||
`artifacts/experiments/collective_commitment_gold_v0/20260820_qwen35_9b_single_run/`.
|
||||
|
||||
## EXP-0026 — Topic-oriented Discussion Subject reconstruction V2 prototype
|
||||
|
||||
Date: 2026-08-11
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Repository entry point for the collective-commitment Gold experiment."""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
if str(REPO_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(REPO_ROOT))
|
||||
|
||||
from src.meeting_lab.controlled_semantic_derivation.experiment_collective import main # noqa: E402
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,349 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Isolated collective-commitment Gold reliability experiment."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from .experiment_h import (
|
||||
DEFAULT_ENDPOINT,
|
||||
DEFAULT_MODEL,
|
||||
DerivationValidationError,
|
||||
OBSERVATION_KEYS,
|
||||
call_ollama,
|
||||
)
|
||||
|
||||
|
||||
GOLD_SCHEMA_VERSION = "experimental-collective-commitment-gold-v0"
|
||||
RECOGNITION_KEYS = {"observation_id", "commitment_form", "normalized_action_text"}
|
||||
COMMITMENT_FORMS = {"individual_first_person", "collective_first_person", "none"}
|
||||
FORBIDDEN_LLM_KEYS = {
|
||||
"responsible_person", "responsibility", "responsibility_scope",
|
||||
"requested_actor", "owner", "ownership", "assignee", "status",
|
||||
"established", "action_item", "protocol", "protocol_category", "decision",
|
||||
"unresolved_issue", "confidence", "relation", "relations", "graph",
|
||||
}
|
||||
WEEKDAYS = {
|
||||
"monday": "Montag", "montag": "Montag", "tuesday": "Dienstag",
|
||||
"dienstag": "Dienstag", "wednesday": "Mittwoch", "mittwoch": "Mittwoch",
|
||||
"thursday": "Donnerstag", "donnerstag": "Donnerstag", "friday": "Freitag",
|
||||
"freitag": "Freitag", "saturday": "Samstag", "samstag": "Samstag",
|
||||
"sunday": "Sonntag", "sonntag": "Sonntag",
|
||||
}
|
||||
|
||||
PROMPT_TEMPLATE = """Recognize only the explicit first-person commitment form and concise action meaning in the supplied single V3-style observation.
|
||||
|
||||
Answer only:
|
||||
1. What explicit first-person commitment form is present?
|
||||
- individual_first_person: the speaker explicitly commits themself personally.
|
||||
- collective_first_person: the speaker explicitly commits a "we" group.
|
||||
- none: there is no explicit first-person commitment.
|
||||
2. What is the concise normalized action meaning?
|
||||
|
||||
Tentative possibility is not commitment. Suggestion or recommendation is not commitment. Impersonal necessity is not commitment. Passive future wording is not commitment. Rejection or negation is not positive commitment. Speaker identity does not convert collective "we" into individual commitment.
|
||||
|
||||
Preserve material limitations such as "nur im Technikum", "nur als Versuch", or "nur 20 Meter" in normalized_action_text. Keep normalized action text in the observation language. When commitment_form is not "none", normalized_action_text must be a non-empty string. When commitment_form is "none", normalized_action_text may be a non-empty action meaning or null.
|
||||
|
||||
Do not infer who is responsible. Do not decide whether an action is established. Do not output responsibility, responsibility scope, requested actor, owner, assignee, status, established, Action Item, protocol, confidence, semantic relations, graphs, decisions, or unresolved issues.
|
||||
|
||||
Return exactly this JSON shape and no additional fields:
|
||||
{{
|
||||
"observation_id": "observation ID",
|
||||
"commitment_form": "individual_first_person | collective_first_person | none",
|
||||
"normalized_action_text": "concise action meaning" | null
|
||||
}}
|
||||
|
||||
V3-style observation:
|
||||
{observation_json}
|
||||
"""
|
||||
|
||||
|
||||
def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None:
|
||||
missing = required - value.keys()
|
||||
unknown = value.keys() - required
|
||||
if missing:
|
||||
raise DerivationValidationError(f"{location} missing required keys: {sorted(missing)}")
|
||||
if unknown:
|
||||
raise DerivationValidationError(f"{location} has unknown keys: {sorted(unknown)}")
|
||||
|
||||
|
||||
def _nonempty_text(value: Any, location: str) -> str:
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise DerivationValidationError(f"{location} must be a non-empty string")
|
||||
return value.strip()
|
||||
|
||||
|
||||
def _validate_observations(observations: Any) -> None:
|
||||
if not isinstance(observations, list) or not observations:
|
||||
raise DerivationValidationError("observations must be a non-empty list")
|
||||
seen_observations: set[str] = set()
|
||||
seen_evidence: set[str] = set()
|
||||
for index, observation in enumerate(observations):
|
||||
location = f"observations[{index}]"
|
||||
if not isinstance(observation, dict):
|
||||
raise DerivationValidationError(f"{location} must be an object")
|
||||
_exact_keys(observation, OBSERVATION_KEYS, location)
|
||||
observation_id = _nonempty_text(observation["observation_id"], f"{location}.observation_id")
|
||||
evidence_id = _nonempty_text(observation["evidence_id"], f"{location}.evidence_id")
|
||||
if observation_id in seen_observations or evidence_id in seen_evidence:
|
||||
raise DerivationValidationError("observation and evidence provenance must be unique")
|
||||
seen_observations.add(observation_id)
|
||||
seen_evidence.add(evidence_id)
|
||||
_nonempty_text(observation["content"], f"{location}.content")
|
||||
_nonempty_text(observation["speaker"], f"{location}.speaker")
|
||||
for field in ("named_person", "addressee"):
|
||||
if observation[field] is not None:
|
||||
_nonempty_text(observation[field], f"{location}.{field}")
|
||||
|
||||
|
||||
def load_gold_cases(path: Path) -> list[dict[str, Any]]:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
if not isinstance(data, dict):
|
||||
raise DerivationValidationError("Gold fixture must be an object")
|
||||
_exact_keys(data, {"schema_version", "cases"}, "Gold fixture")
|
||||
if data["schema_version"] != GOLD_SCHEMA_VERSION:
|
||||
raise DerivationValidationError("unexpected Gold fixture schema_version")
|
||||
cases = data["cases"]
|
||||
if not isinstance(cases, list) or not cases:
|
||||
raise DerivationValidationError("Gold fixture cases must be a non-empty list")
|
||||
seen: set[str] = set()
|
||||
for case in cases:
|
||||
_exact_keys(case, {"case_id", "description", "observations", "expected_recognition", "expected_result"}, "Gold case")
|
||||
case_id = _nonempty_text(case["case_id"], "Gold case.case_id")
|
||||
if case_id in seen:
|
||||
raise DerivationValidationError(f"duplicate case ID: {case_id}")
|
||||
seen.add(case_id)
|
||||
_validate_observations(case["observations"])
|
||||
if len(case["observations"]) != 1:
|
||||
raise DerivationValidationError("collective Gold cases require exactly one observation")
|
||||
return cases
|
||||
|
||||
|
||||
def build_prompt(observations: list[dict[str, Any]]) -> str:
|
||||
_validate_observations(observations)
|
||||
if len(observations) != 1:
|
||||
raise DerivationValidationError("collective recognition requires exactly one observation")
|
||||
return PROMPT_TEMPLATE.format(
|
||||
observation_json=json.dumps(observations[0], ensure_ascii=False, indent=2)
|
||||
)
|
||||
|
||||
|
||||
def parse_model_json(raw_text: str) -> dict[str, Any]:
|
||||
data = json.loads(raw_text)
|
||||
if not isinstance(data, dict):
|
||||
raise DerivationValidationError("semantic recognition must be an object")
|
||||
return data
|
||||
|
||||
|
||||
def _reject_forbidden_keys(value: Any, location: str = "output") -> None:
|
||||
if isinstance(value, dict):
|
||||
forbidden = FORBIDDEN_LLM_KEYS.intersection(value)
|
||||
if forbidden:
|
||||
raise DerivationValidationError(
|
||||
f"{location} contains forbidden semantic keys: {sorted(forbidden)}"
|
||||
)
|
||||
for key, item in value.items():
|
||||
_reject_forbidden_keys(item, f"{location}.{key}")
|
||||
elif isinstance(value, list):
|
||||
for index, item in enumerate(value):
|
||||
_reject_forbidden_keys(item, f"{location}[{index}]")
|
||||
|
||||
|
||||
def validate_recognition(data: Any, observations: list[dict[str, Any]]) -> dict[str, Any]:
|
||||
_validate_observations(observations)
|
||||
if not isinstance(data, dict):
|
||||
raise DerivationValidationError("semantic recognition must be an object")
|
||||
_reject_forbidden_keys(data)
|
||||
_exact_keys(data, RECOGNITION_KEYS, "output")
|
||||
observation_id = _nonempty_text(data["observation_id"], "output.observation_id")
|
||||
if observation_id not in {item["observation_id"] for item in observations}:
|
||||
raise DerivationValidationError("recognition references unknown observation")
|
||||
if data["commitment_form"] not in COMMITMENT_FORMS:
|
||||
raise DerivationValidationError("commitment_form has an unsupported value")
|
||||
action_text = data["normalized_action_text"]
|
||||
if action_text is not None:
|
||||
_nonempty_text(action_text, "output.normalized_action_text")
|
||||
if data["commitment_form"] != "none" and action_text is None:
|
||||
raise DerivationValidationError("non-none commitment requires normalized_action_text")
|
||||
return data
|
||||
|
||||
|
||||
def _bounded_due(observations: list[dict[str, Any]]) -> tuple[str | None, bool]:
|
||||
due_forms: set[str] = set()
|
||||
for observation in observations:
|
||||
content = observation["content"].casefold()
|
||||
for token in re.findall(r"\b[A-Za-zÄÖÜäöü]+\b", content):
|
||||
if token in WEEKDAYS:
|
||||
due_forms.add(WEEKDAYS[token])
|
||||
if re.search(r"\bnächste\s+woche\b", content):
|
||||
due_forms.add("nächste Woche")
|
||||
return (next(iter(due_forms)) if len(due_forms) == 1 else None, len(due_forms) <= 1)
|
||||
|
||||
|
||||
def _has_explicit_negation(observations: list[dict[str, Any]]) -> bool:
|
||||
return any(
|
||||
re.search(r"\b(?:nicht|kein(?:e|en|er|es)?|nein|not|no)\b", item["content"], re.IGNORECASE)
|
||||
for item in observations
|
||||
)
|
||||
|
||||
|
||||
def derive_collective_action(
|
||||
observations: list[dict[str, Any]], recognition: dict[str, Any]
|
||||
) -> tuple[dict[str, bool], dict[str, Any] | None]:
|
||||
validate_recognition(recognition, observations)
|
||||
by_id = {item["observation_id"]: item for item in observations}
|
||||
observation = by_id.get(recognition["observation_id"])
|
||||
due, deadline_consistent = _bounded_due(observations)
|
||||
gates = {
|
||||
"recognition_schema_valid": True,
|
||||
"observation_exists": observation is not None,
|
||||
"provenance_valid_and_unique": observation is not None and len({item["evidence_id"] for item in observations}) == len(observations),
|
||||
"collective_commitment_form": recognition["commitment_form"] == "collective_first_person",
|
||||
"normalized_action_present": isinstance(recognition["normalized_action_text"], str) and bool(recognition["normalized_action_text"].strip()),
|
||||
"deadline_supported_and_consistent": deadline_consistent,
|
||||
"no_explicit_negation": not _has_explicit_negation(observations),
|
||||
}
|
||||
if not all(gates.values()):
|
||||
return gates, None
|
||||
return gates, {
|
||||
"action_id": "action_1",
|
||||
"content": recognition["normalized_action_text"].strip(),
|
||||
"status": "established",
|
||||
"commitment_scope": "collective",
|
||||
"responsible_person": None,
|
||||
"due": due,
|
||||
"support": {
|
||||
"commitment": {
|
||||
"observation_id": observation["observation_id"],
|
||||
"evidence_id": observation["evidence_id"],
|
||||
}
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _concepts_present(text: str | None, concepts: list[list[str]]) -> bool:
|
||||
if not concepts:
|
||||
return True
|
||||
if not isinstance(text, str):
|
||||
return False
|
||||
folded = text.casefold()
|
||||
return all(any(alias.casefold() in folded for alias in alternatives) for alternatives in concepts)
|
||||
|
||||
|
||||
def evaluate_case(case: dict[str, Any], recognition: dict[str, Any]) -> dict[str, Any]:
|
||||
validate_recognition(recognition, case["observations"])
|
||||
gates, result = derive_collective_action(case["observations"], recognition)
|
||||
expected_recognition = case["expected_recognition"]
|
||||
expected_result = case["expected_result"]
|
||||
form_correct = recognition["commitment_form"] == expected_recognition["commitment_form"]
|
||||
action_correct = _concepts_present(recognition["normalized_action_text"], expected_recognition["action_concepts"])
|
||||
qualifier_preserved = _concepts_present(recognition["normalized_action_text"], expected_recognition["qualifier_concepts"])
|
||||
established = result is not None
|
||||
final_correct = established == expected_result["established"]
|
||||
if result is not None:
|
||||
final_correct = final_correct and result["due"] == expected_result["due"] and result["responsible_person"] is None and result["commitment_scope"] == "collective"
|
||||
owner_correct = result is None or result["responsible_person"] is None
|
||||
automatic_failure = (established and not expected_result["established"]) or not owner_correct or (established and not qualifier_preserved)
|
||||
semantic_correct = form_correct and action_correct and qualifier_preserved
|
||||
classification = "FAIL" if automatic_failure or not final_correct else ("PASS" if semantic_correct else "PARTIAL")
|
||||
return {
|
||||
"case_id": case["case_id"], "classification": classification,
|
||||
"commitment_form_correct": form_correct,
|
||||
"normalized_action_meaning_correct": action_correct,
|
||||
"material_qualifier_preserved": qualifier_preserved,
|
||||
"deterministic_gates_correct": final_correct,
|
||||
"final_result_correct": final_correct,
|
||||
"responsible_person_correctly_null": owner_correct,
|
||||
"due_correct": result is None or result["due"] == expected_result["due"],
|
||||
"unsupported_semantic_strengthening": recognition["commitment_form"] == "collective_first_person" and expected_recognition["commitment_form"] != "collective_first_person",
|
||||
"responsibility_status_leakage": False,
|
||||
"gates": gates, "result": result,
|
||||
}
|
||||
|
||||
|
||||
def _write_json(path: Path, value: Any) -> None:
|
||||
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def run_gold(args: argparse.Namespace) -> dict[str, Any]:
|
||||
cases = load_gold_cases(args.cases)
|
||||
args.output.mkdir(parents=True, exist_ok=False)
|
||||
_write_json(args.output / "gold_cases.json", {"schema_version": GOLD_SCHEMA_VERSION, "cases": cases})
|
||||
evaluations: list[dict[str, Any]] = []
|
||||
successful_calls = 0
|
||||
technical_failures = 0
|
||||
started = time.perf_counter()
|
||||
for case in cases:
|
||||
case_dir = args.output / case["case_id"].lower()
|
||||
case_dir.mkdir()
|
||||
observations = case["observations"]
|
||||
_write_json(case_dir / "v3_style_input_observations.json", observations)
|
||||
prompt = build_prompt(observations)
|
||||
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
|
||||
try:
|
||||
raw, metadata = call_ollama(args.endpoint, args.model, prompt, args.timeout, args.num_ctx, args.num_predict)
|
||||
successful_calls += 1
|
||||
except Exception as exc: # one recorded attempt; never retry
|
||||
technical_failures += 1
|
||||
failure = {"case_id": case["case_id"], "classification": "FAIL", "technical_failure": True, "error_type": type(exc).__name__, "error": str(exc)}
|
||||
_write_json(case_dir / "ollama_metadata.json", {"model": args.model, "configuration": {"temperature": 0, "think": False, "num_ctx": args.num_ctx, "num_predict": args.num_predict, "retries": 0}, "technical_failure": failure})
|
||||
_write_json(case_dir / "structural_validation.json", {"valid": False, "error": str(exc)})
|
||||
_write_json(case_dir / "deterministic_gate_results.json", {})
|
||||
_write_json(case_dir / "final_derived_result.json", None)
|
||||
_write_json(case_dir / "evaluation.json", failure)
|
||||
evaluations.append(failure)
|
||||
continue
|
||||
(case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8")
|
||||
_write_json(case_dir / "ollama_metadata.json", metadata)
|
||||
try:
|
||||
parsed = parse_model_json(raw)
|
||||
_write_json(case_dir / "parsed_semantic_recognition.json", parsed)
|
||||
evaluation = evaluate_case(case, parsed)
|
||||
validation = {"valid": True, "error": None}
|
||||
gates, result = derive_collective_action(observations, parsed)
|
||||
except (DerivationValidationError, json.JSONDecodeError) as exc:
|
||||
validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)}
|
||||
evaluation = {"case_id": case["case_id"], "classification": "FAIL", "error": str(exc), "responsibility_status_leakage": "forbidden" in str(exc)}
|
||||
gates, result = {}, None
|
||||
_write_json(case_dir / "structural_validation.json", validation)
|
||||
_write_json(case_dir / "deterministic_gate_results.json", gates)
|
||||
_write_json(case_dir / "final_derived_result.json", result)
|
||||
_write_json(case_dir / "evaluation.json", evaluation)
|
||||
evaluations.append(evaluation)
|
||||
summary = {
|
||||
"experiment": "collective_commitment_gold_v0", "model": args.model,
|
||||
"successful_llm_call_count": successful_calls,
|
||||
"technical_failed_call_count": technical_failures,
|
||||
"runtime_seconds": round(time.perf_counter() - started, 3),
|
||||
"counts": {label: sum(item["classification"] == label for item in evaluations) for label in ("PASS", "PARTIAL", "FAIL")},
|
||||
"evaluations": evaluations,
|
||||
}
|
||||
_write_json(args.output / "summary.json", summary)
|
||||
return summary
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description="Run isolated collective-commitment Gold experiment")
|
||||
parser.add_argument("cases", type=Path)
|
||||
parser.add_argument("-o", "--output", type=Path, required=True)
|
||||
parser.add_argument("--model", default=DEFAULT_MODEL)
|
||||
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
|
||||
parser.add_argument("--timeout", type=int, default=300)
|
||||
parser.add_argument("--num-ctx", type=int, default=16384)
|
||||
parser.add_argument("--num-predict", type=int, default=1024)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
summary = run_gold(parse_args())
|
||||
print(json.dumps(summary, ensure_ascii=False, indent=2))
|
||||
return 0 if summary["counts"]["FAIL"] == 0 and summary["technical_failed_call_count"] == 0 else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,95 @@
|
||||
{
|
||||
"schema_version": "experimental-collective-commitment-gold-v0",
|
||||
"cases": [
|
||||
{
|
||||
"case_id": "CC-01",
|
||||
"description": "Explicit collective commitment",
|
||||
"observations": [
|
||||
{"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Ja, wir testen nächste Woche 20 Meter.", "speaker": "Martin", "named_person": null, "addressee": null}
|
||||
],
|
||||
"expected_recognition": {"commitment_form": "collective_first_person", "action_concepts": [["20"], ["meter", "metre"]], "qualifier_concepts": []},
|
||||
"expected_result": {"established": true, "due": "nächste Woche"}
|
||||
},
|
||||
{
|
||||
"case_id": "CC-02",
|
||||
"description": "Individual commitment",
|
||||
"observations": [
|
||||
{"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Ja, ich teste nächste Woche 20 Meter.", "speaker": "Martin", "named_person": null, "addressee": null}
|
||||
],
|
||||
"expected_recognition": {"commitment_form": "individual_first_person", "action_concepts": [["20"], ["meter", "metre"]], "qualifier_concepts": []},
|
||||
"expected_result": {"established": false, "due": null}
|
||||
},
|
||||
{
|
||||
"case_id": "CC-03",
|
||||
"description": "Tentative collective possibility",
|
||||
"observations": [
|
||||
{"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Wir könnten nächste Woche 20 Meter testen.", "speaker": "Martin", "named_person": null, "addressee": null}
|
||||
],
|
||||
"expected_recognition": {"commitment_form": "none", "action_concepts": [["20"], ["meter", "metre"]], "qualifier_concepts": []},
|
||||
"expected_result": {"established": false, "due": null}
|
||||
},
|
||||
{
|
||||
"case_id": "CC-04",
|
||||
"description": "Collective suggestion",
|
||||
"observations": [
|
||||
{"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Vielleicht sollten wir nächste Woche 20 Meter testen.", "speaker": "Martin", "named_person": null, "addressee": null}
|
||||
],
|
||||
"expected_recognition": {"commitment_form": "none", "action_concepts": [["20"], ["meter", "metre"]], "qualifier_concepts": []},
|
||||
"expected_result": {"established": false, "due": null}
|
||||
},
|
||||
{
|
||||
"case_id": "CC-05",
|
||||
"description": "Impersonal necessity",
|
||||
"observations": [
|
||||
{"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Man müsste nächste Woche 20 Meter testen.", "speaker": "Martin", "named_person": null, "addressee": null}
|
||||
],
|
||||
"expected_recognition": {"commitment_form": "none", "action_concepts": [["20"], ["meter", "metre"]], "qualifier_concepts": []},
|
||||
"expected_result": {"established": false, "due": null}
|
||||
},
|
||||
{
|
||||
"case_id": "CC-06",
|
||||
"description": "Passive future statement",
|
||||
"observations": [
|
||||
{"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Nächste Woche werden 20 Meter getestet.", "speaker": "Martin", "named_person": null, "addressee": null}
|
||||
],
|
||||
"expected_recognition": {"commitment_form": "none", "action_concepts": [["20"], ["meter", "metre"]], "qualifier_concepts": []},
|
||||
"expected_result": {"established": false, "due": null}
|
||||
},
|
||||
{
|
||||
"case_id": "CC-07",
|
||||
"description": "Collective rejection",
|
||||
"observations": [
|
||||
{"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Nein, das testen wir nächste Woche nicht.", "speaker": "Martin", "named_person": null, "addressee": null}
|
||||
],
|
||||
"expected_recognition": {"commitment_form": "none", "action_concepts": [], "qualifier_concepts": []},
|
||||
"expected_result": {"established": false, "due": null}
|
||||
},
|
||||
{
|
||||
"case_id": "CC-08",
|
||||
"description": "Collective commitment with qualifier",
|
||||
"observations": [
|
||||
{"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Ja, wir testen 20 Meter, aber nur im Technikum.", "speaker": "Martin", "named_person": null, "addressee": null}
|
||||
],
|
||||
"expected_recognition": {"commitment_form": "collective_first_person", "action_concepts": [["20"], ["meter", "metre"]], "qualifier_concepts": [["nur", "only"], ["technikum", "technical facility", "technical center", "technical centre"]]},
|
||||
"expected_result": {"established": true, "due": null}
|
||||
},
|
||||
{
|
||||
"case_id": "CC-09",
|
||||
"description": "Collective commitment without deadline",
|
||||
"observations": [
|
||||
{"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Ja, wir testen 20 Meter.", "speaker": "Martin", "named_person": null, "addressee": null}
|
||||
],
|
||||
"expected_recognition": {"commitment_form": "collective_first_person", "action_concepts": [["20"], ["meter", "metre"]], "qualifier_concepts": []},
|
||||
"expected_result": {"established": true, "due": null}
|
||||
},
|
||||
{
|
||||
"case_id": "CC-10",
|
||||
"description": "Speaker ownership trap",
|
||||
"observations": [
|
||||
{"observation_id": "obs_1", "evidence_id": "e1", "content": "Martin: Ja, wir testen nächste Woche 20 Meter.", "speaker": "Martin", "named_person": null, "addressee": null}
|
||||
],
|
||||
"expected_recognition": {"commitment_form": "collective_first_person", "action_concepts": [["20"], ["meter", "metre"]], "qualifier_concepts": []},
|
||||
"expected_result": {"established": true, "due": "nächste Woche"}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,200 @@
|
||||
import json
|
||||
import tempfile
|
||||
import unittest
|
||||
from copy import deepcopy
|
||||
from pathlib import Path
|
||||
|
||||
from src.meeting_lab.controlled_semantic_derivation.experiment_collective import (
|
||||
DerivationValidationError,
|
||||
build_prompt,
|
||||
derive_collective_action,
|
||||
evaluate_case,
|
||||
load_gold_cases,
|
||||
validate_recognition,
|
||||
)
|
||||
|
||||
|
||||
GOLD_PATH = Path("tests/gold/collective_commitment_v0/cases.json")
|
||||
|
||||
|
||||
def recognition_for(case):
|
||||
form = case["expected_recognition"]["commitment_form"]
|
||||
action = "20 Meter testen"
|
||||
if case["case_id"] == "CC-08":
|
||||
action = "20 Meter testen, aber nur im Technikum"
|
||||
return {
|
||||
"observation_id": "obs_1",
|
||||
"commitment_form": form,
|
||||
"normalized_action_text": action,
|
||||
}
|
||||
|
||||
|
||||
class CollectiveCommitmentGoldExperimentTests(unittest.TestCase):
|
||||
@classmethod
|
||||
def setUpClass(cls):
|
||||
cls.cases = load_gold_cases(GOLD_PATH)
|
||||
cls.by_id = {case["case_id"]: case for case in cls.cases}
|
||||
|
||||
def test_fixture_contains_exactly_cc_01_through_cc_10(self):
|
||||
self.assertEqual(list(self.by_id), [f"CC-{number:02d}" for number in range(1, 11)])
|
||||
|
||||
def test_all_cases_are_single_minimal_v3_style_observations(self):
|
||||
keys = {"observation_id", "evidence_id", "content", "speaker", "named_person", "addressee"}
|
||||
for case in self.cases:
|
||||
with self.subTest(case=case["case_id"]):
|
||||
self.assertEqual(len(case["observations"]), 1)
|
||||
self.assertEqual(set(case["observations"][0]), keys)
|
||||
|
||||
def test_cc_01_establishes_collective_action_without_person_and_with_due(self):
|
||||
case = self.by_id["CC-01"]
|
||||
gates, result = derive_collective_action(case["observations"], recognition_for(case))
|
||||
self.assertTrue(all(gates.values()))
|
||||
self.assertEqual(result["status"], "established")
|
||||
self.assertEqual(result["commitment_scope"], "collective")
|
||||
self.assertIsNone(result["responsible_person"])
|
||||
self.assertEqual(result["due"], "nächste Woche")
|
||||
|
||||
def test_individual_commitment_routes_out_of_collective_path(self):
|
||||
self._assert_unestablished("CC-02", "collective_commitment_form")
|
||||
|
||||
def test_tentative_suggestion_impersonal_and_passive_remain_unestablished(self):
|
||||
for case_id in ("CC-03", "CC-04", "CC-05", "CC-06"):
|
||||
with self.subTest(case=case_id):
|
||||
self._assert_unestablished(case_id, "collective_commitment_form")
|
||||
|
||||
def test_rejection_remains_unestablished_and_negation_gate_is_negative(self):
|
||||
case = self.by_id["CC-07"]
|
||||
gates, result = derive_collective_action(case["observations"], recognition_for(case))
|
||||
self.assertFalse(gates["collective_commitment_form"])
|
||||
self.assertFalse(gates["no_explicit_negation"])
|
||||
self.assertIsNone(result)
|
||||
|
||||
def test_qualifier_case_establishes_preserves_limit_and_has_no_due(self):
|
||||
case = self.by_id["CC-08"]
|
||||
_, result = derive_collective_action(case["observations"], recognition_for(case))
|
||||
self.assertIsNotNone(result)
|
||||
self.assertIn("nur im Technikum", result["content"])
|
||||
self.assertIsNone(result["due"])
|
||||
self.assertIsNone(result["responsible_person"])
|
||||
|
||||
def test_collective_without_deadline_establishes_with_null_due(self):
|
||||
case = self.by_id["CC-09"]
|
||||
_, result = derive_collective_action(case["observations"], recognition_for(case))
|
||||
self.assertIsNotNone(result)
|
||||
self.assertIsNone(result["due"])
|
||||
|
||||
def test_speaker_ownership_trap_never_assigns_martin(self):
|
||||
case = self.by_id["CC-10"]
|
||||
_, result = derive_collective_action(case["observations"], recognition_for(case))
|
||||
self.assertIsNotNone(result)
|
||||
self.assertEqual(case["observations"][0]["speaker"], "Martin")
|
||||
self.assertIsNone(result["responsible_person"])
|
||||
|
||||
def test_changing_only_speaker_cannot_create_individual_owner(self):
|
||||
case = deepcopy(self.by_id["CC-01"])
|
||||
for speaker in ("Martin", "Clara", "Antonius"):
|
||||
case["observations"][0]["speaker"] = speaker
|
||||
_, result = derive_collective_action(case["observations"], recognition_for(case))
|
||||
with self.subTest(speaker=speaker):
|
||||
self.assertIsNotNone(result)
|
||||
self.assertIsNone(result["responsible_person"])
|
||||
|
||||
def test_none_and_individual_forms_never_establish(self):
|
||||
case = self.by_id["CC-01"]
|
||||
for form in ("none", "individual_first_person"):
|
||||
recognition = recognition_for(case)
|
||||
recognition["commitment_form"] = form
|
||||
_, result = derive_collective_action(case["observations"], recognition)
|
||||
with self.subTest(form=form):
|
||||
self.assertIsNone(result)
|
||||
|
||||
def test_non_none_commitment_requires_action_text(self):
|
||||
case = self.by_id["CC-01"]
|
||||
recognition = recognition_for(case)
|
||||
recognition["normalized_action_text"] = None
|
||||
with self.assertRaisesRegex(DerivationValidationError, "requires normalized_action_text"):
|
||||
validate_recognition(recognition, case["observations"])
|
||||
|
||||
def test_none_commitment_allows_null_action_text_but_never_establishes(self):
|
||||
case = self.by_id["CC-03"]
|
||||
recognition = recognition_for(case)
|
||||
recognition["normalized_action_text"] = None
|
||||
_, result = derive_collective_action(case["observations"], recognition)
|
||||
self.assertIsNone(result)
|
||||
|
||||
def test_unknown_observation_id_is_rejected(self):
|
||||
case = self.by_id["CC-01"]
|
||||
recognition = recognition_for(case)
|
||||
recognition["observation_id"] = "obs_99"
|
||||
with self.assertRaisesRegex(DerivationValidationError, "unknown observation"):
|
||||
validate_recognition(recognition, case["observations"])
|
||||
|
||||
def test_inconsistent_duplicate_evidence_provenance_is_rejected(self):
|
||||
fixture = json.loads(GOLD_PATH.read_text())
|
||||
fixture["cases"][0]["observations"].append({
|
||||
"observation_id": "obs_2", "evidence_id": "e1", "content": "Martin: Zusatz.",
|
||||
"speaker": "Martin", "named_person": None, "addressee": None,
|
||||
})
|
||||
with tempfile.TemporaryDirectory() as temporary:
|
||||
path = Path(temporary) / "cases.json"
|
||||
path.write_text(json.dumps(fixture), encoding="utf-8")
|
||||
with self.assertRaisesRegex(DerivationValidationError, "provenance must be unique"):
|
||||
load_gold_cases(path)
|
||||
|
||||
def test_conflicting_deadlines_prevent_establishment(self):
|
||||
case = deepcopy(self.by_id["CC-01"])
|
||||
case["observations"][0]["content"] += " Bis Mittwoch."
|
||||
gates, result = derive_collective_action(case["observations"], recognition_for(case))
|
||||
self.assertFalse(gates["deadline_supported_and_consistent"])
|
||||
self.assertIsNone(result)
|
||||
|
||||
def test_forbidden_fields_are_rejected_recursively(self):
|
||||
case = self.by_id["CC-01"]
|
||||
forbidden = (
|
||||
"responsible_person", "responsibility", "responsibility_scope",
|
||||
"requested_actor", "owner", "ownership", "assignee", "status",
|
||||
"established", "action_item", "protocol_category", "decision",
|
||||
"unresolved_issue", "confidence", "relation", "relations", "graph",
|
||||
)
|
||||
for field in forbidden:
|
||||
recognition = recognition_for(case)
|
||||
recognition["wrapper"] = {field: "forbidden"}
|
||||
with self.subTest(field=field), self.assertRaisesRegex(DerivationValidationError, "forbidden semantic keys"):
|
||||
validate_recognition(recognition, case["observations"])
|
||||
|
||||
def test_unknown_schema_field_is_rejected(self):
|
||||
case = self.by_id["CC-01"]
|
||||
recognition = recognition_for(case)
|
||||
recognition["explanation"] = "extra"
|
||||
with self.assertRaisesRegex(DerivationValidationError, "unknown keys"):
|
||||
validate_recognition(recognition, case["observations"])
|
||||
|
||||
def test_provenance_survives_and_successes_always_have_null_person(self):
|
||||
for case_id in ("CC-01", "CC-08", "CC-09", "CC-10"):
|
||||
case = self.by_id[case_id]
|
||||
_, result = derive_collective_action(case["observations"], recognition_for(case))
|
||||
with self.subTest(case=case_id):
|
||||
self.assertEqual(result["support"]["commitment"], {"observation_id": "obs_1", "evidence_id": "e1"})
|
||||
self.assertIsNone(result["responsible_person"])
|
||||
|
||||
def test_all_expected_recognitions_have_correct_final_outcome(self):
|
||||
for case in self.cases:
|
||||
evaluation = evaluate_case(case, recognition_for(case))
|
||||
with self.subTest(case=case["case_id"]):
|
||||
self.assertEqual(evaluation["classification"], "PASS")
|
||||
|
||||
def test_prompt_is_fixed_narrow_and_contains_no_gold_expectation(self):
|
||||
prompt = build_prompt(self.by_id["CC-01"]["observations"])
|
||||
self.assertIn("commitment_form", prompt)
|
||||
self.assertNotIn("expected_result", prompt)
|
||||
self.assertNotIn("Who is responsible", prompt)
|
||||
|
||||
def _assert_unestablished(self, case_id, failed_gate):
|
||||
case = self.by_id[case_id]
|
||||
gates, result = derive_collective_action(case["observations"], recognition_for(case))
|
||||
self.assertFalse(gates[failed_gate])
|
||||
self.assertIsNone(result)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user