Files
meeting-lab/src/meeting_lab/evidence_observations/experiment.py
T
admin 18beb3385f Add evidence-near semantic architecture experiments
Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
2026-08-19 15:46:22 +02:00

396 lines
20 KiB
Python

#!/usr/bin/env python3
"""Extract evidence-near observations for a fixed Discussion Subject."""
from __future__ import annotations
import argparse
import json
import re
import time
from pathlib import Path
from typing import Any
import requests
SCHEMA_VERSION = "experimental-evidence-observations-v1"
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
DEFAULT_MODEL = "qwen3.5:9B"
DEFAULT_TIMEOUT = 300
DEFAULT_NUM_CTX = 16384
DEFAULT_NUM_PREDICT = 4096
RELATIONS = {"none", "supports", "opposes", "qualifies", "limits_scope"}
MODALITIES = {
"factual",
"possible",
"suggested",
"interpersonal_request",
"impersonal_necessity",
"information_question",
"committed",
}
TEMPORALITIES = {"existing", "future", "completed", "unspecified"}
EVALUATIONS = {"positive", "negative", "none"}
AGREEMENTS = {"accepted", "rejected", "unclear", "none"}
RESPONSIBILITIES = {"none", "named", "accepted"}
UNCERTAINTIES = {"present", "absent"}
CLARIFICATION_NEEDS = {"explicit", "implicit", "none"}
OBSERVATION_ID_RE = re.compile(r"^obs_[1-9][0-9]*$")
class ObservationValidationError(ValueError):
"""Raised when an experimental fixture or model output is invalid."""
PROMPT_TEMPLATE = """You extract atomic, evidence-near observations for one fixed Discussion Subject.
Stop before protocol interpretation. Never classify anything as an idea, proposal,
objection, decision, action item, or open question. Do not determine protocol
eligibility, reconstruct topics, generate a protocol, or invent missing stages.
Split an evidence unit into multiple observations when it directly contains multiple
propositions. Preserve every observation's source evidence ID. Use concise content in
the evidence language.
Return exactly one JSON object with this shape:
{{
"schema_version": "experimental-evidence-observations-v1",
"subject_id": "copy exactly",
"subject": "copy exactly",
"observations": [
{{
"observation_id": "obs_1",
"evidence_id": "e1",
"content": "directly supported atomic observation",
"target": "discussion_subject",
"relation": "none",
"modality": "factual",
"temporality": "existing",
"evaluation": "none",
"agreement": "none",
"responsibility": "none",
"person": null,
"uncertainty": "absent",
"clarification_need": "none",
"scope": "absent"
}}
]
}}
Rules:
- Number observation_id sequentially as obs_1, obs_2, ... in evidence order.
- target is "discussion_subject", one earlier observation_id, or a non-empty list of
earlier observation_ids only when the evidence jointly refers to them.
- relation is only none, supports, opposes, qualifies, or limits_scope.
- modality is only factual, possible, suggested, interpersonal_request,
impersonal_necessity, information_question, or committed.
- interpersonal_request is a direct request to another person.
- impersonal_necessity says something needs to happen without assigning it.
- information_question expresses missing information without assigning work.
- temporality is only existing, future, completed, or unspecified.
- evaluation is positive, negative, or none. Do not infer evaluation from world
knowledge. A bare cost or technical fact normally has evaluation none.
- agreement is only accepted, rejected, unclear, or none and applies to target.
- responsibility is none, named, or accepted. Use named only for an explicitly
addressed candidate and accepted only for explicit acceptance/commitment.
- person is the explicit person's name for named/accepted responsibility; otherwise
use JSON null. Mentioning or speaking in first person does not establish ownership.
- uncertainty is present or absent.
- clarification_need is explicit, implicit, or none.
- scope is an evidence-grounded qualifier, or exactly "absent". Never use null or the
string "null" anywhere.
- Confirmation of a rejection targets the rejection observation, not the option.
- A trial-only qualification targets and limits the accepted trial.
- A negative consequence can oppose another observation without requiring
clarification.
- Personal preference is not group rejection.
- Collective "we" does not name an individual owner.
Fixed Gold input:
{input_json}
"""
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(description="Run the evidence-observation Gold experiment.")
parser.add_argument("fixture", type=Path)
parser.add_argument("-o", "--output", type=Path, required=True)
parser.add_argument("--model", default=DEFAULT_MODEL)
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT)
parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX)
parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT)
return parser.parse_args()
def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None:
missing = required - value.keys()
unknown = value.keys() - required
if missing:
raise ObservationValidationError(f"{location} missing required keys: {sorted(missing)}")
if unknown:
raise ObservationValidationError(f"{location} has unknown keys: {sorted(unknown)}")
def _text(value: Any, location: str) -> str:
if not isinstance(value, str) or not value.strip():
raise ObservationValidationError(f"{location} must be a non-empty string")
result = value.strip()
if result.casefold() == "null":
raise ObservationValidationError(f"{location} must not be the string 'null'")
return result
OBSERVATION_KEYS = {
"observation_id", "evidence_id", "content", "target", "relation", "modality",
"temporality", "evaluation", "agreement", "responsibility", "person",
"uncertainty", "clarification_need", "scope",
}
def _validate_target(value: Any, location: str, earlier: set[str]) -> None:
if isinstance(value, str):
target = _text(value, location)
if target != "discussion_subject" and target not in earlier:
raise ObservationValidationError(f"{location} references unknown or later observation: {target}")
return
if not isinstance(value, list) or not value:
raise ObservationValidationError(f"{location} must be discussion_subject, an earlier observation ID, or a non-empty list")
if len(value) < 2:
raise ObservationValidationError(f"{location} list must contain at least two jointly referenced observations")
seen: set[str] = set()
for index, item in enumerate(value):
target = _text(item, f"{location}[{index}]")
if target not in earlier:
raise ObservationValidationError(f"{location}[{index}] references unknown or later observation: {target}")
if target in seen:
raise ObservationValidationError(f"{location} contains duplicate target: {target}")
seen.add(target)
def validate_observations(data: Any, case: dict[str, Any]) -> dict[str, Any]:
validate_case(case)
if not isinstance(data, dict):
raise ObservationValidationError("output must be an object")
_exact_keys(data, {"schema_version", "subject_id", "subject", "observations"}, "output")
if data["schema_version"] != SCHEMA_VERSION:
raise ObservationValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
if data["subject_id"] != case["subject_id"] or data["subject"] != case["subject"]:
raise ObservationValidationError("model changed the fixed Discussion Subject")
observations = data["observations"]
if not isinstance(observations, list) or not observations:
raise ObservationValidationError("output.observations must be a non-empty array")
known_evidence = {item["evidence_id"] for item in case["evidence"]}
earlier: set[str] = set()
for index, observation in enumerate(observations, start=1):
location = f"output.observations[{index - 1}]"
if not isinstance(observation, dict):
raise ObservationValidationError(f"{location} must be an object")
_exact_keys(observation, OBSERVATION_KEYS, location)
observation_id = _text(observation["observation_id"], f"{location}.observation_id")
if not OBSERVATION_ID_RE.fullmatch(observation_id) or observation_id != f"obs_{index}":
raise ObservationValidationError(f"{location}.observation_id must be obs_{index}")
evidence_id = _text(observation["evidence_id"], f"{location}.evidence_id")
if evidence_id not in known_evidence:
raise ObservationValidationError(f"{location}.evidence_id references unknown evidence: {evidence_id}")
_text(observation["content"], f"{location}.content")
_validate_target(observation["target"], f"{location}.target", earlier)
for field, values in (
("relation", RELATIONS), ("modality", MODALITIES),
("temporality", TEMPORALITIES), ("evaluation", EVALUATIONS),
("agreement", AGREEMENTS), ("responsibility", RESPONSIBILITIES),
("uncertainty", UNCERTAINTIES), ("clarification_need", CLARIFICATION_NEEDS),
):
if observation[field] not in values:
raise ObservationValidationError(f"{location}.{field} is invalid: {observation[field]!r}")
person = observation["person"]
if observation["responsibility"] == "none":
if person is not None:
raise ObservationValidationError(f"{location}.person must be JSON null when responsibility is none")
else:
_text(person, f"{location}.person")
scope = _text(observation["scope"], f"{location}.scope")
if scope.casefold() == "null":
raise ObservationValidationError(f"{location}.scope must use 'absent', not 'null'")
earlier.add(observation_id)
return data
def validate_case(case: Any) -> dict[str, Any]:
if not isinstance(case, dict):
raise ObservationValidationError("case must be an object")
_exact_keys(case, {"case_id", "description", "subject_id", "subject", "evidence", "expected_observations"}, "case")
for field in ("case_id", "description", "subject_id", "subject"):
_text(case[field], f"case.{field}")
evidence = case["evidence"]
if not isinstance(evidence, list) or not evidence:
raise ObservationValidationError("case.evidence must be a non-empty array")
seen: set[str] = set()
for index, unit in enumerate(evidence):
location = f"case.evidence[{index}]"
if not isinstance(unit, dict):
raise ObservationValidationError(f"{location} must be an object")
_exact_keys(unit, {"evidence_id", "text"}, location)
evidence_id = _text(unit["evidence_id"], f"{location}.evidence_id")
if evidence_id in seen:
raise ObservationValidationError(f"duplicate evidence ID: {evidence_id}")
seen.add(evidence_id)
_text(unit["text"], f"{location}.text")
expected = case["expected_observations"]
if not isinstance(expected, list) or not expected:
raise ObservationValidationError("case.expected_observations must be a non-empty array")
return case
def validate_fixture_case(case: dict[str, Any]) -> dict[str, Any]:
validate_case(case)
data = {"schema_version": SCHEMA_VERSION, "subject_id": case["subject_id"], "subject": case["subject"], "observations": case["expected_observations"]}
validate_observations(data, case)
return case
def build_prompt(case: dict[str, Any]) -> str:
validate_fixture_case(case)
model_input = {"subject_id": case["subject_id"], "subject": case["subject"], "evidence": case["evidence"]}
return PROMPT_TEMPLATE.format(input_json=json.dumps(model_input, ensure_ascii=False, indent=2))
def parse_model_json(raw_text: str) -> dict[str, Any]:
data = json.loads(raw_text)
if not isinstance(data, dict):
raise ObservationValidationError("model response JSON must be an object")
return data
def build_ollama_payload(model: str, prompt: str, num_ctx: int, num_predict: int) -> dict[str, Any]:
return {"model": model, "prompt": prompt, "think": False, "stream": False, "format": "json", "options": {"temperature": 0, "num_ctx": num_ctx, "num_predict": num_predict}}
def call_ollama(endpoint: str, model: str, prompt: str, timeout: int, num_ctx: int, num_predict: int) -> tuple[str, dict[str, Any]]:
started = time.perf_counter()
response = requests.post(endpoint, json=build_ollama_payload(model, prompt, num_ctx, num_predict), timeout=timeout)
elapsed = time.perf_counter() - started
response.raise_for_status()
body = response.json()
raw_text = body.get("response") if isinstance(body, dict) else None
if not isinstance(raw_text, str) or not raw_text.strip():
raise ValueError("Ollama returned no usable response text")
metadata = {
"model": body.get("model", model), "elapsed_seconds": round(elapsed, 3),
"total_duration_ns": body.get("total_duration"), "load_duration_ns": body.get("load_duration"),
"prompt_eval_count": body.get("prompt_eval_count"), "prompt_eval_duration_ns": body.get("prompt_eval_duration"),
"eval_count": body.get("eval_count"), "eval_duration_ns": body.get("eval_duration"),
"configuration": {"temperature": 0, "think": False, "num_ctx": num_ctx, "num_predict": num_predict, "retries": 0},
}
return raw_text.strip(), metadata
COMPARE_FIELDS = ("evidence_id", "target", "relation", "modality", "temporality", "evaluation", "agreement", "responsibility", "person", "uncertainty", "clarification_need")
def _scope_matches(actual: str, expected: str) -> bool:
if expected == "absent":
return actual == "absent"
expected_terms = [term.strip().casefold() for term in expected.split("|")]
folded = actual.casefold()
return any(term in folded for term in expected_terms)
def evaluate_observations(data: dict[str, Any], expected: list[dict[str, Any]]) -> dict[str, Any]:
actual = data["observations"]
checks: list[dict[str, Any]] = []
pair_count = min(len(actual), len(expected))
checks.append({"name": "observation_count", "passed": len(actual) == len(expected), "critical": False})
categories = {"missing_observations": max(0, len(expected) - len(actual)), "invented_observations": max(0, len(actual) - len(expected)), "stronger_commitment": 0, "weaker_commitment": 0, "incorrect_targets_relations": 0, "incorrect_responsibility": 0, "incorrect_uncertainty_clarification": 0}
commitment_rank = {"factual": 0, "possible": 1, "suggested": 1, "information_question": 1, "impersonal_necessity": 2, "interpersonal_request": 2, "committed": 3}
for index in range(pair_count):
got, want = actual[index], expected[index]
for field in COMPARE_FIELDS:
passed = got[field] == want[field]
checks.append({"name": f"obs_{index + 1}:{field}", "passed": passed, "critical": field in {"evidence_id", "target", "relation", "modality", "agreement", "responsibility", "person"}})
if not passed:
if field in {"target", "relation"}: categories["incorrect_targets_relations"] += 1
if field in {"responsibility", "person"}: categories["incorrect_responsibility"] += 1
if field in {"uncertainty", "clarification_need"}: categories["incorrect_uncertainty_clarification"] += 1
scope_ok = _scope_matches(got["scope"], want["scope"])
checks.append({"name": f"obs_{index + 1}:scope", "passed": scope_ok, "critical": False})
got_rank, want_rank = commitment_rank[got["modality"]], commitment_rank[want["modality"]]
if got_rank > want_rank or (want["agreement"] == "none" and got["agreement"] in {"accepted", "rejected"}): categories["stronger_commitment"] += 1
if got_rank < want_rank or (want["agreement"] in {"accepted", "rejected"} and got["agreement"] == "none"): categories["weaker_commitment"] += 1
passed_count = sum(check["passed"] for check in checks)
critical_failures = [check["name"] for check in checks if check["critical"] and not check["passed"]]
ratio = passed_count / len(checks)
if ratio == 1:
verdict = "PASS"
elif ratio >= 0.7 and categories["stronger_commitment"] == 0 and categories["incorrect_responsibility"] == 0:
verdict = "PARTIAL"
else:
verdict = "FAIL"
return {"verdict": verdict, "matched_checks": passed_count, "check_count": len(checks), "match_ratio": round(ratio, 3), "critical_failures": critical_failures, "error_categories": categories, "checks": checks}
def load_fixture(path: Path) -> list[dict[str, Any]]:
data = json.loads(path.read_text(encoding="utf-8-sig"))
if not isinstance(data, dict) or set(data) != {"cases"} or not isinstance(data["cases"], list) or not data["cases"]:
raise ObservationValidationError("fixture must contain exactly one non-empty cases list")
seen: set[str] = set()
for case in data["cases"]:
validate_fixture_case(case)
if case["case_id"] in seen:
raise ObservationValidationError(f"duplicate case ID: {case['case_id']}")
seen.add(case["case_id"])
return data["cases"]
def _write_json(path: Path, value: Any) -> None:
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
def run_case(case: dict[str, Any], output_root: Path, endpoint: str, model: str, timeout: int, num_ctx: int, num_predict: int) -> dict[str, Any]:
case_dir = output_root / case["case_id"]
case_dir.mkdir(parents=True, exist_ok=False)
_write_json(case_dir / "gold_input.json", {key: case[key] for key in ("case_id", "description", "subject_id", "subject", "evidence")})
_write_json(case_dir / "gold_expected_observations.json", case["expected_observations"])
prompt = build_prompt(case)
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
started = time.perf_counter()
try:
raw, metadata = call_ollama(endpoint, model, prompt, timeout, num_ctx, num_predict)
(case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8")
_write_json(case_dir / "ollama_metadata.json", metadata)
parsed = parse_model_json(raw)
_write_json(case_dir / "parsed_observations.json", parsed)
validate_observations(parsed, case)
validation = {"valid": True, "error": None}
evaluation = evaluate_observations(parsed, case["expected_observations"])
except requests.RequestException:
raise
except (json.JSONDecodeError, ObservationValidationError, ValueError) as exc:
validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)}
evaluation = {"verdict": "FAIL", "matched_checks": 0, "check_count": 0, "match_ratio": 0, "critical_failures": ["schema_validation"], "error_categories": {}, "checks": []}
_write_json(case_dir / "validation_result.json", validation)
result = {"case_id": case["case_id"], **evaluation, "elapsed_seconds": round(time.perf_counter() - started, 3)}
_write_json(case_dir / "evaluation_result.json", result)
return result
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
cases = load_fixture(args.fixture)
args.output.mkdir(parents=True, exist_ok=False)
started = time.perf_counter()
results = []
for index, case in enumerate(cases, start=1):
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
results.append(run_case(case, args.output, args.endpoint, args.model, args.timeout, args.num_ctx, args.num_predict))
summary = {"experiment": "evidence_near_observation_extraction", "schema_version": SCHEMA_VERSION, "model": args.model, "temperature": 0, "think": False, "retries": 0, "case_count": len(cases), "llm_call_count": len(results), "runtime_seconds": round(time.perf_counter() - started, 3), "verdict_counts": {v: sum(r["verdict"] == v for r in results) for v in ("PASS", "PARTIAL", "FAIL")}, "results": results}
_write_json(args.output / "summary.json", summary)
return summary
def main() -> int:
args = parse_args()
summary = run_experiment(args)
print(json.dumps(summary, ensure_ascii=False, indent=2))
return 0 if summary["verdict_counts"]["FAIL"] == 0 else 1