Add controlled request-acceptance derivation experiment
This commit is contained in:
@@ -1627,6 +1627,59 @@ structural validity. This conclusion applies only to the minimal V3 first
|
|||||||
stage and does not determine model choice for any later semantic derivation.
|
stage and does not determine model choice for any later semantic derivation.
|
||||||
No production integration, derivation implementation or Progeo run occurred.
|
No production integration, derivation implementation or Progeo run occurred.
|
||||||
|
|
||||||
|
## EXP-0031 — Controlled Semantic Derivation H V0
|
||||||
|
|
||||||
|
Date: 2026-08-19
|
||||||
|
|
||||||
|
This isolated experiment tested the first controlled second-stage derivation
|
||||||
|
using only the accepted `qwen3.5:9B` V3 observations for case H. The derivation
|
||||||
|
LLM received the two V3 observations, not the transcript or Gold expectation.
|
||||||
|
Its deliberately narrow task was limited to recognizing whether `obs_1` is a
|
||||||
|
concrete request and whether `obs_2` explicitly commits its speaker to
|
||||||
|
substantially the same work. Its strict output schema forbids responsibility,
|
||||||
|
requested actor, establishment/status, Action Item, protocol, confidence and
|
||||||
|
generic relation/graph fields.
|
||||||
|
|
||||||
|
Deterministic code validates observation/evidence provenance, obtains the
|
||||||
|
requested actor only from the request observation's addressee, requires the
|
||||||
|
acceptance to follow the request, requires the accepting speaker to equal that
|
||||||
|
addressee, and establishes responsibility only after all semantic and
|
||||||
|
structural gates pass. A bounded weekday normalizer reconciles `Friday` and
|
||||||
|
`Freitag`, rejects conflicting weekdays, and separates the supported due date
|
||||||
|
from the normalized action text. No general temporal or action ontology was
|
||||||
|
introduced.
|
||||||
|
|
||||||
|
Twenty focused deterministic tests cover the positive H path and the required
|
||||||
|
negative invariants: request alone, acknowledgement/non-commitment, tentative
|
||||||
|
acceptance, different response speaker, different work, reversed order,
|
||||||
|
speaker/name/addressee alone, conflicting deadlines, unknown observation IDs,
|
||||||
|
inconsistent evidence provenance, forbidden semantic fields, malformed JSON
|
||||||
|
and persistent artifacts. The complete non-LLM suite passed 192/192.
|
||||||
|
|
||||||
|
Configuration: one `qwen3.5:9B` call, temperature 0, `think=false`,
|
||||||
|
`num_ctx=16384`, `num_predict=1024`, no retries or voting. The call took 11.765
|
||||||
|
seconds, with 469 prompt-evaluation and 124 evaluation tokens. The model
|
||||||
|
returned a valid recognition object: `obs_1` is a concrete request, `obs_2` is
|
||||||
|
an explicit commitment, and both concern substantially the same work. It
|
||||||
|
returned no responsibility or establishment judgment.
|
||||||
|
|
||||||
|
All deterministic gates passed. The final derived result is an established
|
||||||
|
action `Prüfung der Messdaten`, requested from and assigned to Nina, due
|
||||||
|
`Freitag`, supported by request `obs_1/e1` and acceptance `obs_2/e2`. The model
|
||||||
|
included `bis Friday` in its normalized request text; after the single call, a
|
||||||
|
deterministic-only bounded correction separated that already-recognized due
|
||||||
|
phrase from action content without changing the prompt, recognition schema,
|
||||||
|
semantic result or call count. Focused and complete non-LLM suites still
|
||||||
|
passed after this correction.
|
||||||
|
|
||||||
|
Artifacts are preserved under
|
||||||
|
`artifacts/experiments/controlled_semantic_derivation_h/20260819_h_qwen35_9b_single_run/`.
|
||||||
|
Result: the H mechanism succeeded. This establishes only that the narrow
|
||||||
|
request-plus-explicit-acceptance pattern can be recognized and gated for H; it
|
||||||
|
does not generalize the derivation architecture to other cases or semantic
|
||||||
|
categories. No production integration, other case run, semantic graph,
|
||||||
|
protocol derivation or Progeo run occurred.
|
||||||
|
|
||||||
## EXP-0026 — Topic-oriented Discussion Subject reconstruction V2 prototype
|
## EXP-0026 — Topic-oriented Discussion Subject reconstruction V2 prototype
|
||||||
|
|
||||||
Date: 2026-08-11
|
Date: 2026-08-11
|
||||||
|
|||||||
@@ -0,0 +1,16 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Repository entry point for the H-only controlled derivation experiment."""
|
||||||
|
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||||
|
if str(REPO_ROOT) not in sys.path:
|
||||||
|
sys.path.insert(0, str(REPO_ROOT))
|
||||||
|
|
||||||
|
from src.meeting_lab.controlled_semantic_derivation.experiment_h import main # noqa: E402
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
"""Isolated controlled semantic derivation experiments."""
|
||||||
@@ -0,0 +1,320 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""H-only request/acceptance recognition and deterministic action derivation."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
import time
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import requests
|
||||||
|
|
||||||
|
|
||||||
|
SCHEMA_VERSION = "experimental-controlled-semantic-recognition-h-v0"
|
||||||
|
DEFAULT_MODEL = "qwen3.5:9B"
|
||||||
|
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
|
||||||
|
EXPECTED_PROVENANCE = {"obs_1": "e1", "obs_2": "e2"}
|
||||||
|
OBSERVATION_KEYS = {"observation_id", "evidence_id", "content", "speaker", "named_person", "addressee"}
|
||||||
|
SEMANTIC_KEYS = {"schema_version", "request", "acceptance"}
|
||||||
|
REQUEST_KEYS = {"observation_id", "is_concrete_request", "normalized_action_text"}
|
||||||
|
ACCEPTANCE_KEYS = {"observation_id", "is_explicit_commitment", "same_requested_work", "normalized_action_text"}
|
||||||
|
FORBIDDEN_LLM_KEYS = {
|
||||||
|
"responsible_person", "responsibility", "requested_actor", "status", "established",
|
||||||
|
"action_item", "protocol_section", "protocol_category", "confidence", "relation",
|
||||||
|
"relations", "graph", "decision", "open_question", "unresolved_issue",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class DerivationValidationError(ValueError):
|
||||||
|
"""Raised when experiment input or LLM recognition violates the contract."""
|
||||||
|
|
||||||
|
|
||||||
|
PROMPT_TEMPLATE = """Recognize only two narrow semantic facts in the supplied V3 observations.
|
||||||
|
|
||||||
|
The input contains V3 observations only, not a transcript. Answer only:
|
||||||
|
1. Is obs_1 a concrete request directed to its recorded addressee?
|
||||||
|
2. Does obs_2 explicitly commit its speaker to substantially the same requested work?
|
||||||
|
|
||||||
|
Lexical identity is not required. Conversational paraphrases such as "Prüfung der
|
||||||
|
Messdaten" and "die Prüfung" may denote the same work when the supplied observation
|
||||||
|
sequence clearly supports that reading.
|
||||||
|
|
||||||
|
Do not decide or output responsibility, requested actor, established status, Action
|
||||||
|
Item status, protocol eligibility, confidence, semantic relations, or graphs. Do not
|
||||||
|
answer who is responsible. Deterministic code will apply those gates later.
|
||||||
|
|
||||||
|
Return exactly this JSON shape and no other fields:
|
||||||
|
{{
|
||||||
|
"schema_version": "experimental-controlled-semantic-recognition-h-v0",
|
||||||
|
"request": {{
|
||||||
|
"observation_id": "obs_1",
|
||||||
|
"is_concrete_request": true,
|
||||||
|
"normalized_action_text": "concise requested work in the observation language"
|
||||||
|
}},
|
||||||
|
"acceptance": {{
|
||||||
|
"observation_id": "obs_2",
|
||||||
|
"is_explicit_commitment": true,
|
||||||
|
"same_requested_work": true,
|
||||||
|
"normalized_action_text": "concise accepted work in the observation language"
|
||||||
|
}}
|
||||||
|
}}
|
||||||
|
|
||||||
|
V3 observations:
|
||||||
|
{observations_json}
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
def parse_args() -> argparse.Namespace:
|
||||||
|
parser = argparse.ArgumentParser(description="Run the H-only controlled semantic derivation experiment.")
|
||||||
|
parser.add_argument("observations", type=Path)
|
||||||
|
parser.add_argument("-o", "--output", type=Path, required=True)
|
||||||
|
parser.add_argument("--model", default=DEFAULT_MODEL)
|
||||||
|
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
|
||||||
|
parser.add_argument("--timeout", type=int, default=300)
|
||||||
|
parser.add_argument("--num-ctx", type=int, default=16384)
|
||||||
|
parser.add_argument("--num-predict", type=int, default=1024)
|
||||||
|
return parser.parse_args()
|
||||||
|
|
||||||
|
|
||||||
|
def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None:
|
||||||
|
missing, unknown = required - value.keys(), value.keys() - required
|
||||||
|
if missing:
|
||||||
|
raise DerivationValidationError(f"{location} missing required keys: {sorted(missing)}")
|
||||||
|
if unknown:
|
||||||
|
raise DerivationValidationError(f"{location} has unknown keys: {sorted(unknown)}")
|
||||||
|
|
||||||
|
|
||||||
|
def _text(value: Any, location: str) -> str:
|
||||||
|
if not isinstance(value, str) or not value.strip():
|
||||||
|
raise DerivationValidationError(f"{location} must be a non-empty string")
|
||||||
|
result = value.strip()
|
||||||
|
if result.casefold() == "null":
|
||||||
|
raise DerivationValidationError(f"{location} must not be the string 'null'")
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def load_v3_observations(path: Path) -> list[dict[str, Any]]:
|
||||||
|
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||||
|
if not isinstance(data, dict):
|
||||||
|
raise DerivationValidationError("V3 input must be an object")
|
||||||
|
_exact_keys(data, {"schema_version", "subject_id", "subject", "observations"}, "V3 input")
|
||||||
|
if data["schema_version"] != "experimental-evidence-observations-v3":
|
||||||
|
raise DerivationValidationError("V3 input has an unexpected schema_version")
|
||||||
|
observations = data["observations"]
|
||||||
|
if not isinstance(observations, list) or not observations:
|
||||||
|
raise DerivationValidationError("V3 observations must be a non-empty list")
|
||||||
|
seen: set[str] = set()
|
||||||
|
for index, observation in enumerate(observations):
|
||||||
|
location = f"V3 observations[{index}]"
|
||||||
|
if not isinstance(observation, dict):
|
||||||
|
raise DerivationValidationError(f"{location} must be an object")
|
||||||
|
_exact_keys(observation, OBSERVATION_KEYS, location)
|
||||||
|
observation_id = _text(observation["observation_id"], f"{location}.observation_id")
|
||||||
|
if observation_id in seen:
|
||||||
|
raise DerivationValidationError(f"duplicate observation ID: {observation_id}")
|
||||||
|
seen.add(observation_id)
|
||||||
|
evidence_id = _text(observation["evidence_id"], f"{location}.evidence_id")
|
||||||
|
if observation_id not in EXPECTED_PROVENANCE:
|
||||||
|
raise DerivationValidationError(f"unknown H observation ID: {observation_id}")
|
||||||
|
if EXPECTED_PROVENANCE[observation_id] != evidence_id:
|
||||||
|
raise DerivationValidationError(f"inconsistent evidence provenance for {observation_id}")
|
||||||
|
_text(observation["content"], f"{location}.content")
|
||||||
|
_text(observation["speaker"], f"{location}.speaker")
|
||||||
|
for field in ("named_person", "addressee"):
|
||||||
|
if observation[field] is not None:
|
||||||
|
_text(observation[field], f"{location}.{field}")
|
||||||
|
if seen != set(EXPECTED_PROVENANCE):
|
||||||
|
raise DerivationValidationError("H input must contain exactly obs_1/e1 and obs_2/e2")
|
||||||
|
return observations
|
||||||
|
|
||||||
|
|
||||||
|
def build_prompt(observations: list[dict[str, Any]]) -> str:
|
||||||
|
validate_observation_sequence(observations)
|
||||||
|
return PROMPT_TEMPLATE.format(observations_json=json.dumps(observations, ensure_ascii=False, indent=2))
|
||||||
|
|
||||||
|
|
||||||
|
def validate_observation_sequence(observations: list[dict[str, Any]]) -> None:
|
||||||
|
if [item.get("observation_id") for item in observations] != ["obs_1", "obs_2"]:
|
||||||
|
raise DerivationValidationError("H observations must be ordered obs_1, obs_2")
|
||||||
|
for observation in observations:
|
||||||
|
if EXPECTED_PROVENANCE.get(observation.get("observation_id")) != observation.get("evidence_id"):
|
||||||
|
raise DerivationValidationError("H observation provenance is inconsistent")
|
||||||
|
|
||||||
|
|
||||||
|
def parse_model_json(raw_text: str) -> dict[str, Any]:
|
||||||
|
data = json.loads(raw_text)
|
||||||
|
if not isinstance(data, dict):
|
||||||
|
raise DerivationValidationError("semantic recognition must be an object")
|
||||||
|
return data
|
||||||
|
|
||||||
|
|
||||||
|
def _reject_forbidden_keys(value: Any, location: str = "output") -> None:
|
||||||
|
if isinstance(value, dict):
|
||||||
|
forbidden = FORBIDDEN_LLM_KEYS.intersection(value)
|
||||||
|
if forbidden:
|
||||||
|
raise DerivationValidationError(f"{location} contains forbidden semantic keys: {sorted(forbidden)}")
|
||||||
|
for key, item in value.items():
|
||||||
|
_reject_forbidden_keys(item, f"{location}.{key}")
|
||||||
|
elif isinstance(value, list):
|
||||||
|
for index, item in enumerate(value):
|
||||||
|
_reject_forbidden_keys(item, f"{location}[{index}]")
|
||||||
|
|
||||||
|
|
||||||
|
def validate_semantic_recognition(data: Any, observations: list[dict[str, Any]]) -> dict[str, Any]:
|
||||||
|
if not isinstance(data, dict):
|
||||||
|
raise DerivationValidationError("semantic recognition must be an object")
|
||||||
|
_reject_forbidden_keys(data)
|
||||||
|
_exact_keys(data, SEMANTIC_KEYS, "output")
|
||||||
|
if data["schema_version"] != SCHEMA_VERSION:
|
||||||
|
raise DerivationValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
|
||||||
|
request, acceptance = data["request"], data["acceptance"]
|
||||||
|
if not isinstance(request, dict) or not isinstance(acceptance, dict):
|
||||||
|
raise DerivationValidationError("request and acceptance must be objects")
|
||||||
|
_exact_keys(request, REQUEST_KEYS, "output.request")
|
||||||
|
_exact_keys(acceptance, ACCEPTANCE_KEYS, "output.acceptance")
|
||||||
|
known_ids = {item["observation_id"] for item in observations}
|
||||||
|
for location, item in (("output.request", request), ("output.acceptance", acceptance)):
|
||||||
|
observation_id = _text(item["observation_id"], f"{location}.observation_id")
|
||||||
|
if observation_id not in known_ids:
|
||||||
|
raise DerivationValidationError(f"{location} references unknown observation: {observation_id}")
|
||||||
|
_text(item["normalized_action_text"], f"{location}.normalized_action_text")
|
||||||
|
for field, value in (
|
||||||
|
("output.request.is_concrete_request", request["is_concrete_request"]),
|
||||||
|
("output.acceptance.is_explicit_commitment", acceptance["is_explicit_commitment"]),
|
||||||
|
("output.acceptance.same_requested_work", acceptance["same_requested_work"]),
|
||||||
|
):
|
||||||
|
if not isinstance(value, bool):
|
||||||
|
raise DerivationValidationError(f"{field} must be boolean")
|
||||||
|
if request["observation_id"] == acceptance["observation_id"]:
|
||||||
|
raise DerivationValidationError("request and acceptance must reference different observations")
|
||||||
|
return data
|
||||||
|
|
||||||
|
|
||||||
|
def _bounded_due(observations: list[dict[str, Any]]) -> tuple[str | None, bool]:
|
||||||
|
weekday_forms = {
|
||||||
|
"monday": "Montag", "montag": "Montag",
|
||||||
|
"tuesday": "Dienstag", "dienstag": "Dienstag",
|
||||||
|
"wednesday": "Mittwoch", "mittwoch": "Mittwoch",
|
||||||
|
"thursday": "Donnerstag", "donnerstag": "Donnerstag",
|
||||||
|
"friday": "Freitag", "freitag": "Freitag",
|
||||||
|
"saturday": "Samstag", "samstag": "Samstag",
|
||||||
|
"sunday": "Sonntag", "sonntag": "Sonntag",
|
||||||
|
}
|
||||||
|
forms: set[str] = set()
|
||||||
|
for observation in observations:
|
||||||
|
for token in re.findall(r"\b[A-Za-zÄÖÜäöü]+\b", observation["content"].casefold()):
|
||||||
|
if token in weekday_forms:
|
||||||
|
forms.add(weekday_forms[token])
|
||||||
|
return (next(iter(forms)) if len(forms) == 1 else None, len(forms) <= 1)
|
||||||
|
|
||||||
|
|
||||||
|
def _remove_bounded_due_from_action(action_text: str) -> str:
|
||||||
|
result = re.sub(
|
||||||
|
r"\s+(?:bis|by)\s+(?:Friday|Freitag)\b", "", action_text,
|
||||||
|
flags=re.IGNORECASE,
|
||||||
|
).strip(" .,:;-")
|
||||||
|
return result or action_text.strip()
|
||||||
|
|
||||||
|
|
||||||
|
def derive_action(
|
||||||
|
observations: list[dict[str, Any]], recognition: dict[str, Any]
|
||||||
|
) -> tuple[dict[str, bool], dict[str, Any] | None]:
|
||||||
|
by_id = {item["observation_id"]: item for item in observations}
|
||||||
|
positions = {item["observation_id"]: index for index, item in enumerate(observations)}
|
||||||
|
request_semantic = recognition["request"]
|
||||||
|
acceptance_semantic = recognition["acceptance"]
|
||||||
|
request = by_id.get(request_semantic["observation_id"])
|
||||||
|
acceptance = by_id.get(acceptance_semantic["observation_id"])
|
||||||
|
due, deadline_consistent = _bounded_due(observations)
|
||||||
|
gates = {
|
||||||
|
"request_semantic_positive": request_semantic["is_concrete_request"] is True,
|
||||||
|
"request_observation_exists": request is not None,
|
||||||
|
"request_has_addressee": request is not None and isinstance(request.get("addressee"), str) and bool(request["addressee"].strip()),
|
||||||
|
"acceptance_semantic_positive": acceptance_semantic["is_explicit_commitment"] is True,
|
||||||
|
"same_requested_work": acceptance_semantic["same_requested_work"] is True,
|
||||||
|
"acceptance_observation_exists": acceptance is not None,
|
||||||
|
"acceptance_after_request": request is not None and acceptance is not None and positions[acceptance["observation_id"]] > positions[request["observation_id"]],
|
||||||
|
"acceptance_speaker_matches_addressee": request is not None and acceptance is not None and acceptance["speaker"] == request["addressee"],
|
||||||
|
"provenance_valid_and_consistent": request is not None and acceptance is not None and EXPECTED_PROVENANCE.get(request["observation_id"]) == request["evidence_id"] and EXPECTED_PROVENANCE.get(acceptance["observation_id"]) == acceptance["evidence_id"],
|
||||||
|
"deadline_consistent": deadline_consistent,
|
||||||
|
}
|
||||||
|
if not all(gates.values()):
|
||||||
|
return gates, None
|
||||||
|
action_text = _remove_bounded_due_from_action(
|
||||||
|
request_semantic["normalized_action_text"]
|
||||||
|
)
|
||||||
|
result = {
|
||||||
|
"action_id": "action_1",
|
||||||
|
"content": action_text,
|
||||||
|
"status": "established",
|
||||||
|
"requested_actor": request["addressee"],
|
||||||
|
"responsible_person": acceptance["speaker"],
|
||||||
|
"due": due,
|
||||||
|
"support": {
|
||||||
|
"request": {"observation_id": request["observation_id"], "evidence_id": request["evidence_id"]},
|
||||||
|
"acceptance": {"observation_id": acceptance["observation_id"], "evidence_id": acceptance["evidence_id"]},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
return gates, result
|
||||||
|
|
||||||
|
|
||||||
|
def build_ollama_payload(model: str, prompt: str, num_ctx: int, num_predict: int) -> dict[str, Any]:
|
||||||
|
return {"model": model, "prompt": prompt, "think": False, "stream": False, "format": "json", "options": {"temperature": 0, "num_ctx": num_ctx, "num_predict": num_predict}}
|
||||||
|
|
||||||
|
|
||||||
|
def call_ollama(endpoint: str, model: str, prompt: str, timeout: int, num_ctx: int, num_predict: int) -> tuple[str, dict[str, Any]]:
|
||||||
|
started = time.perf_counter()
|
||||||
|
response = requests.post(endpoint, json=build_ollama_payload(model, prompt, num_ctx, num_predict), timeout=timeout)
|
||||||
|
elapsed = time.perf_counter() - started
|
||||||
|
response.raise_for_status()
|
||||||
|
body = response.json()
|
||||||
|
raw = body.get("response") if isinstance(body, dict) else None
|
||||||
|
if not isinstance(raw, str) or not raw.strip():
|
||||||
|
raise ValueError("Ollama returned no usable response text")
|
||||||
|
metadata = {"model": body.get("model", model), "elapsed_seconds": round(elapsed, 3), "total_duration_ns": body.get("total_duration"), "load_duration_ns": body.get("load_duration"), "prompt_eval_count": body.get("prompt_eval_count"), "prompt_eval_duration_ns": body.get("prompt_eval_duration"), "eval_count": body.get("eval_count"), "eval_duration_ns": body.get("eval_duration"), "configuration": {"temperature": 0, "think": False, "num_ctx": num_ctx, "num_predict": num_predict, "retries": 0}}
|
||||||
|
return raw.strip(), metadata
|
||||||
|
|
||||||
|
|
||||||
|
def _write_json(path: Path, value: Any) -> None:
|
||||||
|
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
|
||||||
|
observations = load_v3_observations(args.observations)
|
||||||
|
args.output.mkdir(parents=True, exist_ok=False)
|
||||||
|
_write_json(args.output / "v3_input_observations.json", observations)
|
||||||
|
prompt = build_prompt(observations)
|
||||||
|
(args.output / "prompt.txt").write_text(prompt, encoding="utf-8")
|
||||||
|
started = time.perf_counter()
|
||||||
|
raw, metadata = call_ollama(args.endpoint, args.model, prompt, args.timeout, args.num_ctx, args.num_predict)
|
||||||
|
(args.output / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8")
|
||||||
|
_write_json(args.output / "ollama_metadata.json", metadata)
|
||||||
|
parsed = parse_model_json(raw)
|
||||||
|
_write_json(args.output / "parsed_semantic_recognition.json", parsed)
|
||||||
|
try:
|
||||||
|
validate_semantic_recognition(parsed, observations)
|
||||||
|
validation = {"valid": True, "error": None}
|
||||||
|
gates, result = derive_action(observations, parsed)
|
||||||
|
except DerivationValidationError as exc:
|
||||||
|
validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)}
|
||||||
|
gates, result = {}, None
|
||||||
|
_write_json(args.output / "structural_validation.json", validation)
|
||||||
|
_write_json(args.output / "deterministic_gate_results.json", gates)
|
||||||
|
_write_json(args.output / "final_derived_result.json", result)
|
||||||
|
summary = {"experiment": "controlled_semantic_derivation_h_v0", "model": args.model, "llm_call_count": 1, "runtime_seconds": round(time.perf_counter() - started, 3), "semantic_recognition_valid": validation["valid"], "all_gates_passed": bool(gates) and all(gates.values()), "action_established": result is not None}
|
||||||
|
_write_json(args.output / "summary.json", summary)
|
||||||
|
return summary
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
args = parse_args()
|
||||||
|
summary = run_experiment(args)
|
||||||
|
print(json.dumps(summary, ensure_ascii=False, indent=2))
|
||||||
|
return 0 if summary["action_established"] else 1
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
raise SystemExit(main())
|
||||||
@@ -0,0 +1,215 @@
|
|||||||
|
import json
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
from copy import deepcopy
|
||||||
|
from pathlib import Path
|
||||||
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
from src.meeting_lab.controlled_semantic_derivation.experiment_h import (
|
||||||
|
SCHEMA_VERSION,
|
||||||
|
DerivationValidationError,
|
||||||
|
build_ollama_payload,
|
||||||
|
derive_action,
|
||||||
|
load_v3_observations,
|
||||||
|
parse_model_json,
|
||||||
|
run_experiment,
|
||||||
|
validate_semantic_recognition,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
ACCEPTED_H_PATH = Path(
|
||||||
|
"artifacts/experiments/evidence_observations_v3/20260819_v3_single_run/"
|
||||||
|
"h_resulting_action/parsed_observations.json"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class ControlledSemanticDerivationHTests(unittest.TestCase):
|
||||||
|
def setUp(self) -> None:
|
||||||
|
self.observations = [
|
||||||
|
{
|
||||||
|
"observation_id": "obs_1", "evidence_id": "e1",
|
||||||
|
"content": "Antonius: Nina, übernimmst du die Prüfung der Messdaten bis Friday?",
|
||||||
|
"speaker": "Antonius", "named_person": "Nina", "addressee": "Nina",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"observation_id": "obs_2", "evidence_id": "e2",
|
||||||
|
"content": "Nina: Ja, ich übernehme die Prüfung bis Freitag.",
|
||||||
|
"speaker": "Nina", "named_person": None, "addressee": None,
|
||||||
|
},
|
||||||
|
]
|
||||||
|
self.recognition = {
|
||||||
|
"schema_version": SCHEMA_VERSION,
|
||||||
|
"request": {
|
||||||
|
"observation_id": "obs_1", "is_concrete_request": True,
|
||||||
|
"normalized_action_text": "Prüfung der Messdaten",
|
||||||
|
},
|
||||||
|
"acceptance": {
|
||||||
|
"observation_id": "obs_2", "is_explicit_commitment": True,
|
||||||
|
"same_requested_work": True,
|
||||||
|
"normalized_action_text": "die Prüfung",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
def derive(self, observations=None, recognition=None):
|
||||||
|
return derive_action(
|
||||||
|
observations if observations is not None else self.observations,
|
||||||
|
recognition if recognition is not None else self.recognition,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_actual_accepted_v3_artifact_is_the_experiment_input(self):
|
||||||
|
actual = load_v3_observations(ACCEPTED_H_PATH)
|
||||||
|
self.assertEqual(actual, self.observations)
|
||||||
|
|
||||||
|
def test_valid_semantic_recognition_has_no_derivation_fields(self):
|
||||||
|
validate_semantic_recognition(self.recognition, self.observations)
|
||||||
|
serialized = json.dumps(self.recognition)
|
||||||
|
for forbidden in ("responsible_person", "requested_actor", "status", "established", "action_item"):
|
||||||
|
self.assertNotIn(forbidden, serialized)
|
||||||
|
|
||||||
|
def test_request_and_acceptance_provenance_survive(self):
|
||||||
|
gates, result = self.derive()
|
||||||
|
self.assertTrue(all(gates.values()))
|
||||||
|
self.assertEqual(result["support"]["request"], {"observation_id": "obs_1", "evidence_id": "e1"})
|
||||||
|
self.assertEqual(result["support"]["acceptance"], {"observation_id": "obs_2", "evidence_id": "e2"})
|
||||||
|
|
||||||
|
def test_valid_sequence_establishes_expected_action(self):
|
||||||
|
recognition = deepcopy(self.recognition)
|
||||||
|
recognition["request"]["normalized_action_text"] = "Prüfung der Messdaten bis Friday"
|
||||||
|
_, result = self.derive(recognition=recognition)
|
||||||
|
self.assertEqual(result["content"], "Prüfung der Messdaten")
|
||||||
|
self.assertEqual(result["status"], "established")
|
||||||
|
self.assertEqual(result["requested_actor"], "Nina")
|
||||||
|
self.assertEqual(result["responsible_person"], "Nina")
|
||||||
|
self.assertEqual(result["due"], "Freitag")
|
||||||
|
|
||||||
|
def test_lexical_identity_is_not_required(self):
|
||||||
|
self.assertNotEqual(
|
||||||
|
self.recognition["request"]["normalized_action_text"],
|
||||||
|
self.recognition["acceptance"]["normalized_action_text"],
|
||||||
|
)
|
||||||
|
gates, result = self.derive()
|
||||||
|
self.assertTrue(gates["same_requested_work"])
|
||||||
|
self.assertIsNotNone(result)
|
||||||
|
|
||||||
|
def test_request_alone_does_not_establish(self):
|
||||||
|
gates, result = self.derive(observations=self.observations[:1])
|
||||||
|
self.assertFalse(gates["acceptance_observation_exists"])
|
||||||
|
self.assertIsNone(result)
|
||||||
|
|
||||||
|
def test_noncommitting_or_acknowledging_response_does_not_establish(self):
|
||||||
|
recognition = deepcopy(self.recognition)
|
||||||
|
recognition["acceptance"]["is_explicit_commitment"] = False
|
||||||
|
gates, result = self.derive(recognition=recognition)
|
||||||
|
self.assertFalse(gates["acceptance_semantic_positive"])
|
||||||
|
self.assertIsNone(result)
|
||||||
|
|
||||||
|
def test_different_response_speaker_does_not_establish(self):
|
||||||
|
observations = deepcopy(self.observations)
|
||||||
|
observations[1]["speaker"] = "Martin"
|
||||||
|
gates, result = self.derive(observations=observations)
|
||||||
|
self.assertFalse(gates["acceptance_speaker_matches_addressee"])
|
||||||
|
self.assertIsNone(result)
|
||||||
|
|
||||||
|
def test_different_accepted_work_does_not_establish(self):
|
||||||
|
recognition = deepcopy(self.recognition)
|
||||||
|
recognition["acceptance"]["same_requested_work"] = False
|
||||||
|
recognition["acceptance"]["normalized_action_text"] = "Angebot prüfen"
|
||||||
|
gates, result = self.derive(recognition=recognition)
|
||||||
|
self.assertFalse(gates["same_requested_work"])
|
||||||
|
self.assertIsNone(result)
|
||||||
|
|
||||||
|
def test_tentative_acceptance_does_not_establish(self):
|
||||||
|
recognition = deepcopy(self.recognition)
|
||||||
|
recognition["acceptance"]["is_explicit_commitment"] = False
|
||||||
|
_, result = self.derive(recognition=recognition)
|
||||||
|
self.assertIsNone(result)
|
||||||
|
|
||||||
|
def test_named_person_speaker_or_addressee_alone_cannot_establish(self):
|
||||||
|
recognition = deepcopy(self.recognition)
|
||||||
|
recognition["acceptance"]["is_explicit_commitment"] = False
|
||||||
|
gates, result = self.derive(recognition=recognition)
|
||||||
|
self.assertEqual(self.observations[0]["named_person"], "Nina")
|
||||||
|
self.assertEqual(self.observations[0]["addressee"], "Nina")
|
||||||
|
self.assertEqual(self.observations[1]["speaker"], "Nina")
|
||||||
|
self.assertFalse(gates["acceptance_semantic_positive"])
|
||||||
|
self.assertIsNone(result)
|
||||||
|
|
||||||
|
def test_acceptance_must_follow_request(self):
|
||||||
|
observations = list(reversed(deepcopy(self.observations)))
|
||||||
|
gates, result = self.derive(observations=observations)
|
||||||
|
self.assertFalse(gates["acceptance_after_request"])
|
||||||
|
self.assertIsNone(result)
|
||||||
|
|
||||||
|
def test_conflicting_deadlines_do_not_establish(self):
|
||||||
|
observations = deepcopy(self.observations)
|
||||||
|
observations[1]["content"] = "Nina: Ja, ich übernehme die Prüfung bis Donnerstag."
|
||||||
|
gates, result = self.derive(observations=observations)
|
||||||
|
self.assertFalse(gates["deadline_consistent"])
|
||||||
|
self.assertIsNone(result)
|
||||||
|
|
||||||
|
def test_unknown_observation_reference_is_rejected(self):
|
||||||
|
recognition = deepcopy(self.recognition)
|
||||||
|
recognition["acceptance"]["observation_id"] = "obs_9"
|
||||||
|
with self.assertRaisesRegex(DerivationValidationError, "unknown observation"):
|
||||||
|
validate_semantic_recognition(recognition, self.observations)
|
||||||
|
|
||||||
|
def test_inconsistent_evidence_provenance_is_rejected(self):
|
||||||
|
data = json.loads(ACCEPTED_H_PATH.read_text())
|
||||||
|
data["observations"][1]["evidence_id"] = "e1"
|
||||||
|
with tempfile.TemporaryDirectory() as temporary:
|
||||||
|
path = Path(temporary) / "observations.json"
|
||||||
|
path.write_text(json.dumps(data), encoding="utf-8")
|
||||||
|
with self.assertRaisesRegex(DerivationValidationError, "inconsistent evidence provenance"):
|
||||||
|
load_v3_observations(path)
|
||||||
|
|
||||||
|
def test_responsibility_or_status_in_llm_output_is_rejected(self):
|
||||||
|
for field in ("responsibility", "responsible_person", "status", "established"):
|
||||||
|
recognition = deepcopy(self.recognition)
|
||||||
|
recognition[field] = "forbidden"
|
||||||
|
with self.subTest(field=field), self.assertRaisesRegex(DerivationValidationError, "forbidden semantic keys"):
|
||||||
|
validate_semantic_recognition(recognition, self.observations)
|
||||||
|
|
||||||
|
def test_protocol_or_unrelated_semantic_concepts_are_rejected(self):
|
||||||
|
for field in ("protocol_category", "decision", "unresolved_issue", "graph", "confidence"):
|
||||||
|
recognition = deepcopy(self.recognition)
|
||||||
|
recognition[field] = "forbidden"
|
||||||
|
with self.subTest(field=field), self.assertRaisesRegex(DerivationValidationError, "forbidden semantic keys"):
|
||||||
|
validate_semantic_recognition(recognition, self.observations)
|
||||||
|
|
||||||
|
def test_malformed_json_is_rejected(self):
|
||||||
|
with self.assertRaises(json.JSONDecodeError):
|
||||||
|
parse_model_json("{bad json")
|
||||||
|
|
||||||
|
def test_payload_has_one_call_controls(self):
|
||||||
|
payload = build_ollama_payload("qwen3.5:9B", "prompt", 16384, 1024)
|
||||||
|
self.assertFalse(payload["think"])
|
||||||
|
self.assertFalse(payload["stream"])
|
||||||
|
self.assertEqual(payload["options"]["temperature"], 0)
|
||||||
|
|
||||||
|
def test_run_preserves_all_artifacts_without_real_ollama(self):
|
||||||
|
raw = json.dumps(self.recognition, ensure_ascii=False)
|
||||||
|
from argparse import Namespace
|
||||||
|
|
||||||
|
with tempfile.TemporaryDirectory() as temporary:
|
||||||
|
output = Path(temporary) / "run"
|
||||||
|
args = Namespace(
|
||||||
|
observations=ACCEPTED_H_PATH, output=output, model="qwen3.5:9B",
|
||||||
|
endpoint="http://unused", timeout=1, num_ctx=16384, num_predict=1024,
|
||||||
|
)
|
||||||
|
with patch(
|
||||||
|
"src.meeting_lab.controlled_semantic_derivation.experiment_h.call_ollama",
|
||||||
|
return_value=(raw, {"model": "qwen3.5:9B"}),
|
||||||
|
):
|
||||||
|
summary = run_experiment(args)
|
||||||
|
self.assertTrue(summary["action_established"])
|
||||||
|
for filename in (
|
||||||
|
"v3_input_observations.json", "prompt.txt", "raw_model_response.txt",
|
||||||
|
"parsed_semantic_recognition.json", "structural_validation.json",
|
||||||
|
"deterministic_gate_results.json", "final_derived_result.json",
|
||||||
|
"ollama_metadata.json", "summary.json",
|
||||||
|
):
|
||||||
|
self.assertTrue((output / filename).is_file(), filename)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
Reference in New Issue
Block a user