Files
meeting-lab/src/meeting_lab/topic_reconstruction/experiment.py
T
admin 18beb3385f Add evidence-near semantic architecture experiments
Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
2026-08-19 15:46:22 +02:00

722 lines
26 KiB
Python

#!/usr/bin/env python3
"""Run an isolated Discussion Subject reconstruction experiment with Ollama."""
from __future__ import annotations
import argparse
import json
import re
import time
from pathlib import Path
from typing import Any
import requests
SCHEMA_VERSION = "experimental-discussion-subjects-v1"
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
DEFAULT_MODEL = "qwen3.5:9B"
DEFAULT_TIMEOUT = 300
DEFAULT_NUM_CTX = 16384
DEFAULT_NUM_PREDICT = 4096
EVENT_TYPES = {
"introduced_idea",
"considered_option",
"proposal",
"supporting_argument",
"objection",
"clarification",
"modification",
"fact",
"technical_finding",
}
OUTCOME_CERTAINTIES = {"established", "tentative", "conditional", "rejected"}
IDENTIFIER_RE = re.compile(r"^[a-z][a-z0-9_]*$")
class ReconstructionValidationError(ValueError):
"""Raised when experimental reconstruction output violates the schema."""
PROMPT_TEMPLATE = """You reconstruct discussion subjects from meeting evidence.
This is semantic reconstruction, not protocol writing and not flat category extraction.
Group evidence by what participants are actually discussing. For each subject, record
only supported discourse events and, when present, the actual outcome, resulting
actions, and genuinely unresolved issues.
Important distinctions:
- discussed is not necessarily proposed
- proposed is not necessarily preferred or accepted
- preferred is not accepted
- accepted for a trial is not accepted as a final solution
- mentioned is not an unresolved question
- an outcome must preserve its scope, conditions, polarity, and uncertainty
- do not infer responsibility from mention, expertise, adjacency, or likely role
- do not invent missing stages or emit empty optional structures
Evidence discipline:
- Use only the supplied evidence IDs in evidence_refs.
- Every subject, event, outcome, action, and unresolved issue needs at least one
evidence reference.
- Keep statements concise; do not copy long evidence passages.
- A subject may consist only of one introduced idea.
Return one JSON object with exactly:
{{
"schema_version": "experimental-discussion-subjects-v1",
"subjects": [
{{
"subject_id": "subject_1",
"title": "concise discussion subject",
"evidence_refs": ["e1"],
"development": [
{{
"event_id": "event_1",
"type": "introduced_idea|considered_option|proposal|supporting_argument|objection|clarification|modification|fact|technical_finding",
"text": "what happened in the discussion",
"evidence_refs": ["e1"]
}}
],
"outcome": {{
"text": "only what was established",
"scope": "explicit limit or full scope of the outcome",
"certainty": "established|tentative|conditional|rejected",
"evidence_refs": ["e2"]
}},
"actions": [
{{
"action_id": "action_1",
"text": "established work only",
"responsible": "explicitly supported name or null",
"deadline": "explicitly supported deadline or null",
"evidence_refs": ["e3"]
}}
],
"unresolved_issues": [
{{
"issue_id": "issue_1",
"text": "concrete unresolved issue",
"evidence_refs": ["e4"]
}}
]
}}
]
}}
Only subject_id, title, evidence_refs are required for each subject. Omit
development, outcome, actions, or unresolved_issues when absent. Never emit null
or an empty optional list/object.
Case ID: {case_id}
Evidence units:
{evidence_json}
"""
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Run the isolated topic-reconstruction Gold experiment."
)
parser.add_argument("fixture", type=Path, help="Focused Gold cases JSON.")
parser.add_argument("-o", "--output", type=Path, required=True)
parser.add_argument("--model", default=DEFAULT_MODEL)
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT)
parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX)
parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT)
parser.add_argument(
"--case", action="append", dest="case_ids", help="Run only this case ID."
)
return parser.parse_args()
def _expect_exact_keys(
value: dict[str, Any], required: set[str], optional: set[str], location: str
) -> None:
missing = required - value.keys()
unknown = value.keys() - required - optional
if missing:
raise ReconstructionValidationError(
f"{location} missing required keys: {sorted(missing)}"
)
if unknown:
raise ReconstructionValidationError(
f"{location} has unknown keys: {sorted(unknown)}"
)
def _nonempty_text(value: Any, location: str) -> str:
if not isinstance(value, str) or not value.strip():
raise ReconstructionValidationError(f"{location} must be a non-empty string")
return value.strip()
def _identifier(value: Any, location: str, seen: set[str]) -> str:
text = _nonempty_text(value, location)
if not IDENTIFIER_RE.fullmatch(text):
raise ReconstructionValidationError(f"{location} is not a valid identifier")
if text in seen:
raise ReconstructionValidationError(f"duplicate identifier: {text}")
seen.add(text)
return text
def _nullable_text(value: Any, location: str) -> str | None:
if value is None:
return None
text = _nonempty_text(value, location)
if text.casefold() == "null":
raise ReconstructionValidationError(
f"{location} must use JSON null, not the string 'null'"
)
return text
def _evidence_refs(value: Any, location: str, known: set[str]) -> list[str]:
if not isinstance(value, list) or not value:
raise ReconstructionValidationError(f"{location} must be a non-empty list")
refs: list[str] = []
for index, ref in enumerate(value):
ref = _nonempty_text(ref, f"{location}[{index}]")
if ref not in known:
raise ReconstructionValidationError(
f"{location}[{index}] references unknown evidence ID: {ref}"
)
if ref in refs:
raise ReconstructionValidationError(
f"{location} contains duplicate evidence reference: {ref}"
)
refs.append(ref)
return refs
def validate_evidence_units(evidence_units: Any) -> set[str]:
if not isinstance(evidence_units, list) or not evidence_units:
raise ReconstructionValidationError("evidence_units must be a non-empty list")
known: set[str] = set()
for index, unit in enumerate(evidence_units):
location = f"evidence_units[{index}]"
if not isinstance(unit, dict):
raise ReconstructionValidationError(f"{location} must be an object")
_expect_exact_keys(unit, {"evidence_id", "text"}, set(), location)
evidence_id = _nonempty_text(unit["evidence_id"], f"{location}.evidence_id")
if evidence_id in known:
raise ReconstructionValidationError(
f"duplicate input evidence identifier: {evidence_id}"
)
known.add(evidence_id)
_nonempty_text(unit["text"], f"{location}.text")
return known
def validate_reconstruction(data: Any, evidence_units: Any) -> dict[str, Any]:
known = validate_evidence_units(evidence_units)
if not isinstance(data, dict):
raise ReconstructionValidationError("model output must be an object")
_expect_exact_keys(data, {"schema_version", "subjects"}, set(), "output")
if data["schema_version"] != SCHEMA_VERSION:
raise ReconstructionValidationError(
f"schema_version must be {SCHEMA_VERSION!r}"
)
subjects = data["subjects"]
if not isinstance(subjects, list) or not subjects:
raise ReconstructionValidationError("subjects must be a non-empty list")
seen: set[str] = set()
for subject_index, subject in enumerate(subjects):
location = f"subjects[{subject_index}]"
if not isinstance(subject, dict):
raise ReconstructionValidationError(f"{location} must be an object")
_expect_exact_keys(
subject,
{"subject_id", "title", "evidence_refs"},
{"development", "outcome", "actions", "unresolved_issues"},
location,
)
_identifier(subject["subject_id"], f"{location}.subject_id", seen)
_nonempty_text(subject["title"], f"{location}.title")
_evidence_refs(subject["evidence_refs"], f"{location}.evidence_refs", known)
if "development" in subject:
events = subject["development"]
if not isinstance(events, list) or not events:
raise ReconstructionValidationError(
f"{location}.development must be a non-empty list when present"
)
for event_index, event in enumerate(events):
event_location = f"{location}.development[{event_index}]"
if not isinstance(event, dict):
raise ReconstructionValidationError(
f"{event_location} must be an object"
)
_expect_exact_keys(
event,
{"event_id", "type", "text", "evidence_refs"},
set(),
event_location,
)
_identifier(event["event_id"], f"{event_location}.event_id", seen)
if event["type"] not in EVENT_TYPES:
raise ReconstructionValidationError(
f"{event_location}.type is invalid: {event['type']!r}"
)
_nonempty_text(event["text"], f"{event_location}.text")
_evidence_refs(
event["evidence_refs"], f"{event_location}.evidence_refs", known
)
if "outcome" in subject:
outcome = subject["outcome"]
outcome_location = f"{location}.outcome"
if not isinstance(outcome, dict):
raise ReconstructionValidationError(
f"{outcome_location} must be a non-empty object when present"
)
_expect_exact_keys(
outcome,
{"text", "scope", "certainty", "evidence_refs"},
set(),
outcome_location,
)
_nonempty_text(outcome["text"], f"{outcome_location}.text")
_nonempty_text(outcome["scope"], f"{outcome_location}.scope")
if outcome["certainty"] not in OUTCOME_CERTAINTIES:
raise ReconstructionValidationError(
f"{outcome_location}.certainty is invalid: {outcome['certainty']!r}"
)
_evidence_refs(
outcome["evidence_refs"], f"{outcome_location}.evidence_refs", known
)
if "actions" in subject:
actions = subject["actions"]
if not isinstance(actions, list) or not actions:
raise ReconstructionValidationError(
f"{location}.actions must be a non-empty list when present"
)
for action_index, action in enumerate(actions):
action_location = f"{location}.actions[{action_index}]"
if not isinstance(action, dict):
raise ReconstructionValidationError(
f"{action_location} must be an object"
)
_expect_exact_keys(
action,
{"action_id", "text", "responsible", "deadline", "evidence_refs"},
set(),
action_location,
)
_identifier(action["action_id"], f"{action_location}.action_id", seen)
_nonempty_text(action["text"], f"{action_location}.text")
for field in ("responsible", "deadline"):
_nullable_text(action[field], f"{action_location}.{field}")
_evidence_refs(
action["evidence_refs"], f"{action_location}.evidence_refs", known
)
if "unresolved_issues" in subject:
issues = subject["unresolved_issues"]
if not isinstance(issues, list) or not issues:
raise ReconstructionValidationError(
f"{location}.unresolved_issues must be a non-empty list when present"
)
for issue_index, issue in enumerate(issues):
issue_location = f"{location}.unresolved_issues[{issue_index}]"
if not isinstance(issue, dict):
raise ReconstructionValidationError(
f"{issue_location} must be an object"
)
_expect_exact_keys(
issue,
{"issue_id", "text", "evidence_refs"},
set(),
issue_location,
)
_identifier(issue["issue_id"], f"{issue_location}.issue_id", seen)
_nonempty_text(issue["text"], f"{issue_location}.text")
_evidence_refs(
issue["evidence_refs"], f"{issue_location}.evidence_refs", known
)
return data
def build_prompt(case: dict[str, Any]) -> str:
evidence_units = case["evidence_units"]
validate_evidence_units(evidence_units)
return PROMPT_TEMPLATE.format(
case_id=case["case_id"],
evidence_json=json.dumps(evidence_units, ensure_ascii=False, indent=2),
)
def parse_model_json(raw_text: str) -> dict[str, Any]:
data = json.loads(raw_text)
if not isinstance(data, dict):
raise ReconstructionValidationError("model response JSON must be an object")
return data
def build_ollama_payload(
model: str, prompt: str, num_ctx: int, num_predict: int
) -> dict[str, Any]:
return {
"model": model,
"prompt": prompt,
"think": False,
"stream": False,
"format": "json",
"options": {
"temperature": 0,
"num_ctx": num_ctx,
"num_predict": num_predict,
},
}
def call_ollama(
endpoint: str,
model: str,
prompt: str,
timeout: int,
num_ctx: int,
num_predict: int,
) -> tuple[str, dict[str, Any]]:
payload = build_ollama_payload(model, prompt, num_ctx, num_predict)
started = time.perf_counter()
response = requests.post(endpoint, json=payload, timeout=timeout)
elapsed = time.perf_counter() - started
response.raise_for_status()
data = response.json()
if not isinstance(data, dict):
raise ValueError("Ollama response must be a JSON object")
raw_text = data.get("response")
if not isinstance(raw_text, str) or not raw_text.strip():
raise ValueError("Ollama returned no usable response text")
metadata = {
"model": data.get("model", model),
"elapsed_seconds": round(elapsed, 3),
"total_duration_ns": data.get("total_duration"),
"load_duration_ns": data.get("load_duration"),
"prompt_eval_count": data.get("prompt_eval_count"),
"prompt_eval_duration_ns": data.get("prompt_eval_duration"),
"eval_count": data.get("eval_count"),
"eval_duration_ns": data.get("eval_duration"),
"configuration": {
"temperature": 0,
"think": False,
"num_ctx": num_ctx,
"num_predict": num_predict,
},
}
return raw_text.strip(), metadata
def _all_text(subjects: list[dict[str, Any]]) -> str:
parts: list[str] = []
for subject in subjects:
parts.append(subject["title"])
for event in subject.get("development", []):
parts.append(event["text"])
outcome = subject.get("outcome")
if outcome:
parts.extend((outcome["text"], outcome["scope"]))
for action in subject.get("actions", []):
parts.append(action["text"])
for issue in subject.get("unresolved_issues", []):
parts.append(issue["text"])
return " ".join(parts).casefold()
def _contains_any(text: str, terms: list[str]) -> bool:
return any(term.casefold() in text for term in terms)
def evaluate_reconstruction(
reconstruction: dict[str, Any], expected: dict[str, Any]
) -> dict[str, Any]:
subjects = reconstruction["subjects"]
combined = _all_text(subjects)
events = [event for subject in subjects for event in subject.get("development", [])]
outcomes = [subject["outcome"] for subject in subjects if "outcome" in subject]
actions = [action for subject in subjects for action in subject.get("actions", [])]
issues = [issue for subject in subjects for issue in subject.get("unresolved_issues", [])]
checks: list[dict[str, Any]] = []
def add(name: str, passed: bool, critical: bool = False) -> None:
checks.append({"name": name, "passed": passed, "critical": critical})
add("subject_count", len(subjects) == expected.get("subject_count", 1))
add("subject_identity", _contains_any(combined, expected["subject_terms"]))
event_types = {event["type"] for event in events}
for event_type in expected.get("required_event_types", []):
add(f"event_type:{event_type}", event_type in event_types)
expected_outcome = expected.get("outcome", {})
outcome_required = expected_outcome.get("required", False)
add(
"outcome_presence",
bool(outcomes) is outcome_required,
critical=not outcome_required and bool(outcomes),
)
if outcome_required and outcomes:
outcome_text = " ".join(
f"{item['text']} {item['scope']}" for item in outcomes
).casefold()
add("outcome_meaning", _contains_any(outcome_text, expected_outcome["terms"]))
add(
"outcome_scope",
_contains_any(outcome_text, expected_outcome.get("scope_terms", [])),
critical=True,
)
add(
"outcome_certainty",
any(
item["certainty"] in expected_outcome.get("certainties", [])
for item in outcomes
),
)
expected_actions = expected.get("actions", {})
minimum_actions = expected_actions.get("minimum", 0)
add(
"action_count",
len(actions) >= minimum_actions if minimum_actions else not actions,
critical=minimum_actions == 0 and bool(actions),
)
if minimum_actions and actions:
action_text = " ".join(item["text"] for item in actions).casefold()
add("action_meaning", _contains_any(action_text, expected_actions["terms"]))
if "responsible" in expected_actions:
add(
"action_responsibility",
any(
item["responsible"] == expected_actions["responsible"]
for item in actions
),
critical=True,
)
expected_issues = expected.get("unresolved", {})
minimum_issues = expected_issues.get("minimum", 0)
add(
"unresolved_count",
len(issues) >= minimum_issues if minimum_issues else not issues,
critical=minimum_issues == 0 and bool(issues),
)
if minimum_issues and issues:
issue_text = " ".join(item["text"] for item in issues).casefold()
add("unresolved_meaning", _contains_any(issue_text, expected_issues["terms"]))
passed = sum(check["passed"] for check in checks)
critical_failures = [
check["name"] for check in checks if check["critical"] and not check["passed"]
]
ratio = passed / len(checks)
if ratio == 1:
verdict = "PASS"
elif ratio >= 0.6 and not critical_failures:
verdict = "PARTIAL"
else:
verdict = "FAIL"
failed = [check["name"] for check in checks if not check["passed"]]
reason = "All semantic checks passed." if not failed else "Failed: " + ", ".join(failed)
return {
"verdict": verdict,
"reason": reason,
"passed_checks": passed,
"check_count": len(checks),
"critical_failures": critical_failures,
"checks": checks,
}
def load_fixture(path: Path) -> list[dict[str, Any]]:
data = json.loads(path.read_text(encoding="utf-8-sig"))
if not isinstance(data, dict) or set(data) != {"cases"}:
raise ValueError("fixture must contain exactly one 'cases' list")
cases = data["cases"]
if not isinstance(cases, list) or not cases:
raise ValueError("fixture cases must be a non-empty list")
seen: set[str] = set()
for index, case in enumerate(cases):
if not isinstance(case, dict):
raise ValueError(f"cases[{index}] must be an object")
required = {"case_id", "description", "evidence_units", "expected"}
if set(case) != required:
raise ValueError(f"cases[{index}] must contain exactly {sorted(required)}")
case_id = _nonempty_text(case["case_id"], f"cases[{index}].case_id")
if case_id in seen:
raise ValueError(f"duplicate case_id: {case_id}")
seen.add(case_id)
_nonempty_text(case["description"], f"cases[{index}].description")
validate_evidence_units(case["evidence_units"])
if not isinstance(case["expected"], dict):
raise ValueError(f"cases[{index}].expected must be an object")
return cases
def run_case(
case: dict[str, Any],
output_root: Path,
endpoint: str,
model: str,
timeout: int,
num_ctx: int,
num_predict: int,
) -> dict[str, Any]:
case_dir = output_root / case["case_id"]
case_dir.mkdir(parents=True, exist_ok=False)
input_payload = {
"case_id": case["case_id"],
"description": case["description"],
"evidence_units": case["evidence_units"],
}
(case_dir / "input.json").write_text(
json.dumps(input_payload, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
prompt = build_prompt(case)
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
started = time.perf_counter()
try:
raw_text, metadata = call_ollama(
endpoint, model, prompt, timeout, num_ctx, num_predict
)
(case_dir / "raw_model_response.txt").write_text(
raw_text + "\n", encoding="utf-8"
)
(case_dir / "ollama_metadata.json").write_text(
json.dumps(metadata, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
parsed = parse_model_json(raw_text)
(case_dir / "parsed_output.json").write_text(
json.dumps(parsed, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
validated = validate_reconstruction(parsed, case["evidence_units"])
evaluation = evaluate_reconstruction(validated, case["expected"])
except requests.RequestException as exc:
failure = {
"case_id": case["case_id"],
"error_type": type(exc).__name__,
"error": str(exc),
"elapsed_seconds": round(time.perf_counter() - started, 3),
}
(case_dir / "validation_failure.json").write_text(
json.dumps(failure, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
raise
except (json.JSONDecodeError, ReconstructionValidationError, ValueError) as exc:
elapsed = round(time.perf_counter() - started, 3)
failure = {
"case_id": case["case_id"],
"error_type": type(exc).__name__,
"error": str(exc),
"elapsed_seconds": elapsed,
}
(case_dir / "validation_failure.json").write_text(
json.dumps(failure, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
result = {
"case_id": case["case_id"],
"description": case["description"],
"verdict": "FAIL",
"reason": f"{type(exc).__name__}: {exc}",
"passed_checks": 0,
"check_count": 0,
"critical_failures": ["schema_validation"],
"checks": [],
"elapsed_seconds": elapsed,
"subject_titles": [],
}
(case_dir / "evaluation.json").write_text(
json.dumps(result, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
return result
result = {
"case_id": case["case_id"],
"description": case["description"],
**evaluation,
"elapsed_seconds": metadata["elapsed_seconds"],
"subject_titles": [item["title"] for item in validated["subjects"]],
}
(case_dir / "evaluation.json").write_text(
json.dumps(result, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
return result
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
cases = load_fixture(args.fixture)
selected = set(args.case_ids or [])
if selected:
known = {case["case_id"] for case in cases}
unknown = selected - known
if unknown:
raise ValueError(f"unknown requested case IDs: {sorted(unknown)}")
cases = [case for case in cases if case["case_id"] in selected]
args.output.mkdir(parents=True, exist_ok=False)
results: list[dict[str, Any]] = []
started = time.perf_counter()
for index, case in enumerate(cases, start=1):
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
results.append(
run_case(
case,
args.output,
args.endpoint,
args.model,
args.timeout,
args.num_ctx,
args.num_predict,
)
)
summary = {
"experiment": "topic_reconstruction_v2",
"schema_version": SCHEMA_VERSION,
"model": args.model,
"temperature": 0,
"think": False,
"case_count": len(cases),
"llm_call_count": len(results),
"runtime_seconds": round(time.perf_counter() - started, 3),
"verdict_counts": {
verdict: sum(item["verdict"] == verdict for item in results)
for verdict in ("PASS", "PARTIAL", "FAIL")
},
"results": results,
}
(args.output / "summary.json").write_text(
json.dumps(summary, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
return summary
def main() -> int:
args = parse_args()
try:
summary = run_experiment(args)
except (OSError, ValueError, requests.RequestException) as exc:
print(f"Error: {exc}")
return 1
print(json.dumps(summary["verdict_counts"], sort_keys=True))
print(f"Artifacts: {args.output.resolve()}")
return 0
if __name__ == "__main__":
raise SystemExit(main())