Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
722 lines
26 KiB
Python
722 lines
26 KiB
Python
#!/usr/bin/env python3
|
|
"""Run an isolated Discussion Subject reconstruction experiment with Ollama."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import re
|
|
import time
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
import requests
|
|
|
|
|
|
SCHEMA_VERSION = "experimental-discussion-subjects-v1"
|
|
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
|
|
DEFAULT_MODEL = "qwen3.5:9B"
|
|
DEFAULT_TIMEOUT = 300
|
|
DEFAULT_NUM_CTX = 16384
|
|
DEFAULT_NUM_PREDICT = 4096
|
|
|
|
EVENT_TYPES = {
|
|
"introduced_idea",
|
|
"considered_option",
|
|
"proposal",
|
|
"supporting_argument",
|
|
"objection",
|
|
"clarification",
|
|
"modification",
|
|
"fact",
|
|
"technical_finding",
|
|
}
|
|
OUTCOME_CERTAINTIES = {"established", "tentative", "conditional", "rejected"}
|
|
IDENTIFIER_RE = re.compile(r"^[a-z][a-z0-9_]*$")
|
|
|
|
|
|
class ReconstructionValidationError(ValueError):
|
|
"""Raised when experimental reconstruction output violates the schema."""
|
|
|
|
|
|
PROMPT_TEMPLATE = """You reconstruct discussion subjects from meeting evidence.
|
|
|
|
This is semantic reconstruction, not protocol writing and not flat category extraction.
|
|
Group evidence by what participants are actually discussing. For each subject, record
|
|
only supported discourse events and, when present, the actual outcome, resulting
|
|
actions, and genuinely unresolved issues.
|
|
|
|
Important distinctions:
|
|
- discussed is not necessarily proposed
|
|
- proposed is not necessarily preferred or accepted
|
|
- preferred is not accepted
|
|
- accepted for a trial is not accepted as a final solution
|
|
- mentioned is not an unresolved question
|
|
- an outcome must preserve its scope, conditions, polarity, and uncertainty
|
|
- do not infer responsibility from mention, expertise, adjacency, or likely role
|
|
- do not invent missing stages or emit empty optional structures
|
|
|
|
Evidence discipline:
|
|
- Use only the supplied evidence IDs in evidence_refs.
|
|
- Every subject, event, outcome, action, and unresolved issue needs at least one
|
|
evidence reference.
|
|
- Keep statements concise; do not copy long evidence passages.
|
|
- A subject may consist only of one introduced idea.
|
|
|
|
Return one JSON object with exactly:
|
|
{{
|
|
"schema_version": "experimental-discussion-subjects-v1",
|
|
"subjects": [
|
|
{{
|
|
"subject_id": "subject_1",
|
|
"title": "concise discussion subject",
|
|
"evidence_refs": ["e1"],
|
|
"development": [
|
|
{{
|
|
"event_id": "event_1",
|
|
"type": "introduced_idea|considered_option|proposal|supporting_argument|objection|clarification|modification|fact|technical_finding",
|
|
"text": "what happened in the discussion",
|
|
"evidence_refs": ["e1"]
|
|
}}
|
|
],
|
|
"outcome": {{
|
|
"text": "only what was established",
|
|
"scope": "explicit limit or full scope of the outcome",
|
|
"certainty": "established|tentative|conditional|rejected",
|
|
"evidence_refs": ["e2"]
|
|
}},
|
|
"actions": [
|
|
{{
|
|
"action_id": "action_1",
|
|
"text": "established work only",
|
|
"responsible": "explicitly supported name or null",
|
|
"deadline": "explicitly supported deadline or null",
|
|
"evidence_refs": ["e3"]
|
|
}}
|
|
],
|
|
"unresolved_issues": [
|
|
{{
|
|
"issue_id": "issue_1",
|
|
"text": "concrete unresolved issue",
|
|
"evidence_refs": ["e4"]
|
|
}}
|
|
]
|
|
}}
|
|
]
|
|
}}
|
|
|
|
Only subject_id, title, evidence_refs are required for each subject. Omit
|
|
development, outcome, actions, or unresolved_issues when absent. Never emit null
|
|
or an empty optional list/object.
|
|
|
|
Case ID: {case_id}
|
|
Evidence units:
|
|
{evidence_json}
|
|
"""
|
|
|
|
|
|
def parse_args() -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(
|
|
description="Run the isolated topic-reconstruction Gold experiment."
|
|
)
|
|
parser.add_argument("fixture", type=Path, help="Focused Gold cases JSON.")
|
|
parser.add_argument("-o", "--output", type=Path, required=True)
|
|
parser.add_argument("--model", default=DEFAULT_MODEL)
|
|
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
|
|
parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT)
|
|
parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX)
|
|
parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT)
|
|
parser.add_argument(
|
|
"--case", action="append", dest="case_ids", help="Run only this case ID."
|
|
)
|
|
return parser.parse_args()
|
|
|
|
|
|
def _expect_exact_keys(
|
|
value: dict[str, Any], required: set[str], optional: set[str], location: str
|
|
) -> None:
|
|
missing = required - value.keys()
|
|
unknown = value.keys() - required - optional
|
|
if missing:
|
|
raise ReconstructionValidationError(
|
|
f"{location} missing required keys: {sorted(missing)}"
|
|
)
|
|
if unknown:
|
|
raise ReconstructionValidationError(
|
|
f"{location} has unknown keys: {sorted(unknown)}"
|
|
)
|
|
|
|
|
|
def _nonempty_text(value: Any, location: str) -> str:
|
|
if not isinstance(value, str) or not value.strip():
|
|
raise ReconstructionValidationError(f"{location} must be a non-empty string")
|
|
return value.strip()
|
|
|
|
|
|
def _identifier(value: Any, location: str, seen: set[str]) -> str:
|
|
text = _nonempty_text(value, location)
|
|
if not IDENTIFIER_RE.fullmatch(text):
|
|
raise ReconstructionValidationError(f"{location} is not a valid identifier")
|
|
if text in seen:
|
|
raise ReconstructionValidationError(f"duplicate identifier: {text}")
|
|
seen.add(text)
|
|
return text
|
|
|
|
|
|
def _nullable_text(value: Any, location: str) -> str | None:
|
|
if value is None:
|
|
return None
|
|
text = _nonempty_text(value, location)
|
|
if text.casefold() == "null":
|
|
raise ReconstructionValidationError(
|
|
f"{location} must use JSON null, not the string 'null'"
|
|
)
|
|
return text
|
|
|
|
|
|
def _evidence_refs(value: Any, location: str, known: set[str]) -> list[str]:
|
|
if not isinstance(value, list) or not value:
|
|
raise ReconstructionValidationError(f"{location} must be a non-empty list")
|
|
refs: list[str] = []
|
|
for index, ref in enumerate(value):
|
|
ref = _nonempty_text(ref, f"{location}[{index}]")
|
|
if ref not in known:
|
|
raise ReconstructionValidationError(
|
|
f"{location}[{index}] references unknown evidence ID: {ref}"
|
|
)
|
|
if ref in refs:
|
|
raise ReconstructionValidationError(
|
|
f"{location} contains duplicate evidence reference: {ref}"
|
|
)
|
|
refs.append(ref)
|
|
return refs
|
|
|
|
|
|
def validate_evidence_units(evidence_units: Any) -> set[str]:
|
|
if not isinstance(evidence_units, list) or not evidence_units:
|
|
raise ReconstructionValidationError("evidence_units must be a non-empty list")
|
|
known: set[str] = set()
|
|
for index, unit in enumerate(evidence_units):
|
|
location = f"evidence_units[{index}]"
|
|
if not isinstance(unit, dict):
|
|
raise ReconstructionValidationError(f"{location} must be an object")
|
|
_expect_exact_keys(unit, {"evidence_id", "text"}, set(), location)
|
|
evidence_id = _nonempty_text(unit["evidence_id"], f"{location}.evidence_id")
|
|
if evidence_id in known:
|
|
raise ReconstructionValidationError(
|
|
f"duplicate input evidence identifier: {evidence_id}"
|
|
)
|
|
known.add(evidence_id)
|
|
_nonempty_text(unit["text"], f"{location}.text")
|
|
return known
|
|
|
|
|
|
def validate_reconstruction(data: Any, evidence_units: Any) -> dict[str, Any]:
|
|
known = validate_evidence_units(evidence_units)
|
|
if not isinstance(data, dict):
|
|
raise ReconstructionValidationError("model output must be an object")
|
|
_expect_exact_keys(data, {"schema_version", "subjects"}, set(), "output")
|
|
if data["schema_version"] != SCHEMA_VERSION:
|
|
raise ReconstructionValidationError(
|
|
f"schema_version must be {SCHEMA_VERSION!r}"
|
|
)
|
|
subjects = data["subjects"]
|
|
if not isinstance(subjects, list) or not subjects:
|
|
raise ReconstructionValidationError("subjects must be a non-empty list")
|
|
|
|
seen: set[str] = set()
|
|
for subject_index, subject in enumerate(subjects):
|
|
location = f"subjects[{subject_index}]"
|
|
if not isinstance(subject, dict):
|
|
raise ReconstructionValidationError(f"{location} must be an object")
|
|
_expect_exact_keys(
|
|
subject,
|
|
{"subject_id", "title", "evidence_refs"},
|
|
{"development", "outcome", "actions", "unresolved_issues"},
|
|
location,
|
|
)
|
|
_identifier(subject["subject_id"], f"{location}.subject_id", seen)
|
|
_nonempty_text(subject["title"], f"{location}.title")
|
|
_evidence_refs(subject["evidence_refs"], f"{location}.evidence_refs", known)
|
|
|
|
if "development" in subject:
|
|
events = subject["development"]
|
|
if not isinstance(events, list) or not events:
|
|
raise ReconstructionValidationError(
|
|
f"{location}.development must be a non-empty list when present"
|
|
)
|
|
for event_index, event in enumerate(events):
|
|
event_location = f"{location}.development[{event_index}]"
|
|
if not isinstance(event, dict):
|
|
raise ReconstructionValidationError(
|
|
f"{event_location} must be an object"
|
|
)
|
|
_expect_exact_keys(
|
|
event,
|
|
{"event_id", "type", "text", "evidence_refs"},
|
|
set(),
|
|
event_location,
|
|
)
|
|
_identifier(event["event_id"], f"{event_location}.event_id", seen)
|
|
if event["type"] not in EVENT_TYPES:
|
|
raise ReconstructionValidationError(
|
|
f"{event_location}.type is invalid: {event['type']!r}"
|
|
)
|
|
_nonempty_text(event["text"], f"{event_location}.text")
|
|
_evidence_refs(
|
|
event["evidence_refs"], f"{event_location}.evidence_refs", known
|
|
)
|
|
|
|
if "outcome" in subject:
|
|
outcome = subject["outcome"]
|
|
outcome_location = f"{location}.outcome"
|
|
if not isinstance(outcome, dict):
|
|
raise ReconstructionValidationError(
|
|
f"{outcome_location} must be a non-empty object when present"
|
|
)
|
|
_expect_exact_keys(
|
|
outcome,
|
|
{"text", "scope", "certainty", "evidence_refs"},
|
|
set(),
|
|
outcome_location,
|
|
)
|
|
_nonempty_text(outcome["text"], f"{outcome_location}.text")
|
|
_nonempty_text(outcome["scope"], f"{outcome_location}.scope")
|
|
if outcome["certainty"] not in OUTCOME_CERTAINTIES:
|
|
raise ReconstructionValidationError(
|
|
f"{outcome_location}.certainty is invalid: {outcome['certainty']!r}"
|
|
)
|
|
_evidence_refs(
|
|
outcome["evidence_refs"], f"{outcome_location}.evidence_refs", known
|
|
)
|
|
|
|
if "actions" in subject:
|
|
actions = subject["actions"]
|
|
if not isinstance(actions, list) or not actions:
|
|
raise ReconstructionValidationError(
|
|
f"{location}.actions must be a non-empty list when present"
|
|
)
|
|
for action_index, action in enumerate(actions):
|
|
action_location = f"{location}.actions[{action_index}]"
|
|
if not isinstance(action, dict):
|
|
raise ReconstructionValidationError(
|
|
f"{action_location} must be an object"
|
|
)
|
|
_expect_exact_keys(
|
|
action,
|
|
{"action_id", "text", "responsible", "deadline", "evidence_refs"},
|
|
set(),
|
|
action_location,
|
|
)
|
|
_identifier(action["action_id"], f"{action_location}.action_id", seen)
|
|
_nonempty_text(action["text"], f"{action_location}.text")
|
|
for field in ("responsible", "deadline"):
|
|
_nullable_text(action[field], f"{action_location}.{field}")
|
|
_evidence_refs(
|
|
action["evidence_refs"], f"{action_location}.evidence_refs", known
|
|
)
|
|
|
|
if "unresolved_issues" in subject:
|
|
issues = subject["unresolved_issues"]
|
|
if not isinstance(issues, list) or not issues:
|
|
raise ReconstructionValidationError(
|
|
f"{location}.unresolved_issues must be a non-empty list when present"
|
|
)
|
|
for issue_index, issue in enumerate(issues):
|
|
issue_location = f"{location}.unresolved_issues[{issue_index}]"
|
|
if not isinstance(issue, dict):
|
|
raise ReconstructionValidationError(
|
|
f"{issue_location} must be an object"
|
|
)
|
|
_expect_exact_keys(
|
|
issue,
|
|
{"issue_id", "text", "evidence_refs"},
|
|
set(),
|
|
issue_location,
|
|
)
|
|
_identifier(issue["issue_id"], f"{issue_location}.issue_id", seen)
|
|
_nonempty_text(issue["text"], f"{issue_location}.text")
|
|
_evidence_refs(
|
|
issue["evidence_refs"], f"{issue_location}.evidence_refs", known
|
|
)
|
|
|
|
return data
|
|
|
|
|
|
def build_prompt(case: dict[str, Any]) -> str:
|
|
evidence_units = case["evidence_units"]
|
|
validate_evidence_units(evidence_units)
|
|
return PROMPT_TEMPLATE.format(
|
|
case_id=case["case_id"],
|
|
evidence_json=json.dumps(evidence_units, ensure_ascii=False, indent=2),
|
|
)
|
|
|
|
|
|
def parse_model_json(raw_text: str) -> dict[str, Any]:
|
|
data = json.loads(raw_text)
|
|
if not isinstance(data, dict):
|
|
raise ReconstructionValidationError("model response JSON must be an object")
|
|
return data
|
|
|
|
|
|
def build_ollama_payload(
|
|
model: str, prompt: str, num_ctx: int, num_predict: int
|
|
) -> dict[str, Any]:
|
|
return {
|
|
"model": model,
|
|
"prompt": prompt,
|
|
"think": False,
|
|
"stream": False,
|
|
"format": "json",
|
|
"options": {
|
|
"temperature": 0,
|
|
"num_ctx": num_ctx,
|
|
"num_predict": num_predict,
|
|
},
|
|
}
|
|
|
|
|
|
def call_ollama(
|
|
endpoint: str,
|
|
model: str,
|
|
prompt: str,
|
|
timeout: int,
|
|
num_ctx: int,
|
|
num_predict: int,
|
|
) -> tuple[str, dict[str, Any]]:
|
|
payload = build_ollama_payload(model, prompt, num_ctx, num_predict)
|
|
started = time.perf_counter()
|
|
response = requests.post(endpoint, json=payload, timeout=timeout)
|
|
elapsed = time.perf_counter() - started
|
|
response.raise_for_status()
|
|
data = response.json()
|
|
if not isinstance(data, dict):
|
|
raise ValueError("Ollama response must be a JSON object")
|
|
raw_text = data.get("response")
|
|
if not isinstance(raw_text, str) or not raw_text.strip():
|
|
raise ValueError("Ollama returned no usable response text")
|
|
metadata = {
|
|
"model": data.get("model", model),
|
|
"elapsed_seconds": round(elapsed, 3),
|
|
"total_duration_ns": data.get("total_duration"),
|
|
"load_duration_ns": data.get("load_duration"),
|
|
"prompt_eval_count": data.get("prompt_eval_count"),
|
|
"prompt_eval_duration_ns": data.get("prompt_eval_duration"),
|
|
"eval_count": data.get("eval_count"),
|
|
"eval_duration_ns": data.get("eval_duration"),
|
|
"configuration": {
|
|
"temperature": 0,
|
|
"think": False,
|
|
"num_ctx": num_ctx,
|
|
"num_predict": num_predict,
|
|
},
|
|
}
|
|
return raw_text.strip(), metadata
|
|
|
|
|
|
def _all_text(subjects: list[dict[str, Any]]) -> str:
|
|
parts: list[str] = []
|
|
for subject in subjects:
|
|
parts.append(subject["title"])
|
|
for event in subject.get("development", []):
|
|
parts.append(event["text"])
|
|
outcome = subject.get("outcome")
|
|
if outcome:
|
|
parts.extend((outcome["text"], outcome["scope"]))
|
|
for action in subject.get("actions", []):
|
|
parts.append(action["text"])
|
|
for issue in subject.get("unresolved_issues", []):
|
|
parts.append(issue["text"])
|
|
return " ".join(parts).casefold()
|
|
|
|
|
|
def _contains_any(text: str, terms: list[str]) -> bool:
|
|
return any(term.casefold() in text for term in terms)
|
|
|
|
|
|
def evaluate_reconstruction(
|
|
reconstruction: dict[str, Any], expected: dict[str, Any]
|
|
) -> dict[str, Any]:
|
|
subjects = reconstruction["subjects"]
|
|
combined = _all_text(subjects)
|
|
events = [event for subject in subjects for event in subject.get("development", [])]
|
|
outcomes = [subject["outcome"] for subject in subjects if "outcome" in subject]
|
|
actions = [action for subject in subjects for action in subject.get("actions", [])]
|
|
issues = [issue for subject in subjects for issue in subject.get("unresolved_issues", [])]
|
|
checks: list[dict[str, Any]] = []
|
|
|
|
def add(name: str, passed: bool, critical: bool = False) -> None:
|
|
checks.append({"name": name, "passed": passed, "critical": critical})
|
|
|
|
add("subject_count", len(subjects) == expected.get("subject_count", 1))
|
|
add("subject_identity", _contains_any(combined, expected["subject_terms"]))
|
|
|
|
event_types = {event["type"] for event in events}
|
|
for event_type in expected.get("required_event_types", []):
|
|
add(f"event_type:{event_type}", event_type in event_types)
|
|
|
|
expected_outcome = expected.get("outcome", {})
|
|
outcome_required = expected_outcome.get("required", False)
|
|
add(
|
|
"outcome_presence",
|
|
bool(outcomes) is outcome_required,
|
|
critical=not outcome_required and bool(outcomes),
|
|
)
|
|
if outcome_required and outcomes:
|
|
outcome_text = " ".join(
|
|
f"{item['text']} {item['scope']}" for item in outcomes
|
|
).casefold()
|
|
add("outcome_meaning", _contains_any(outcome_text, expected_outcome["terms"]))
|
|
add(
|
|
"outcome_scope",
|
|
_contains_any(outcome_text, expected_outcome.get("scope_terms", [])),
|
|
critical=True,
|
|
)
|
|
add(
|
|
"outcome_certainty",
|
|
any(
|
|
item["certainty"] in expected_outcome.get("certainties", [])
|
|
for item in outcomes
|
|
),
|
|
)
|
|
|
|
expected_actions = expected.get("actions", {})
|
|
minimum_actions = expected_actions.get("minimum", 0)
|
|
add(
|
|
"action_count",
|
|
len(actions) >= minimum_actions if minimum_actions else not actions,
|
|
critical=minimum_actions == 0 and bool(actions),
|
|
)
|
|
if minimum_actions and actions:
|
|
action_text = " ".join(item["text"] for item in actions).casefold()
|
|
add("action_meaning", _contains_any(action_text, expected_actions["terms"]))
|
|
if "responsible" in expected_actions:
|
|
add(
|
|
"action_responsibility",
|
|
any(
|
|
item["responsible"] == expected_actions["responsible"]
|
|
for item in actions
|
|
),
|
|
critical=True,
|
|
)
|
|
|
|
expected_issues = expected.get("unresolved", {})
|
|
minimum_issues = expected_issues.get("minimum", 0)
|
|
add(
|
|
"unresolved_count",
|
|
len(issues) >= minimum_issues if minimum_issues else not issues,
|
|
critical=minimum_issues == 0 and bool(issues),
|
|
)
|
|
if minimum_issues and issues:
|
|
issue_text = " ".join(item["text"] for item in issues).casefold()
|
|
add("unresolved_meaning", _contains_any(issue_text, expected_issues["terms"]))
|
|
|
|
passed = sum(check["passed"] for check in checks)
|
|
critical_failures = [
|
|
check["name"] for check in checks if check["critical"] and not check["passed"]
|
|
]
|
|
ratio = passed / len(checks)
|
|
if ratio == 1:
|
|
verdict = "PASS"
|
|
elif ratio >= 0.6 and not critical_failures:
|
|
verdict = "PARTIAL"
|
|
else:
|
|
verdict = "FAIL"
|
|
failed = [check["name"] for check in checks if not check["passed"]]
|
|
reason = "All semantic checks passed." if not failed else "Failed: " + ", ".join(failed)
|
|
return {
|
|
"verdict": verdict,
|
|
"reason": reason,
|
|
"passed_checks": passed,
|
|
"check_count": len(checks),
|
|
"critical_failures": critical_failures,
|
|
"checks": checks,
|
|
}
|
|
|
|
|
|
def load_fixture(path: Path) -> list[dict[str, Any]]:
|
|
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
|
if not isinstance(data, dict) or set(data) != {"cases"}:
|
|
raise ValueError("fixture must contain exactly one 'cases' list")
|
|
cases = data["cases"]
|
|
if not isinstance(cases, list) or not cases:
|
|
raise ValueError("fixture cases must be a non-empty list")
|
|
seen: set[str] = set()
|
|
for index, case in enumerate(cases):
|
|
if not isinstance(case, dict):
|
|
raise ValueError(f"cases[{index}] must be an object")
|
|
required = {"case_id", "description", "evidence_units", "expected"}
|
|
if set(case) != required:
|
|
raise ValueError(f"cases[{index}] must contain exactly {sorted(required)}")
|
|
case_id = _nonempty_text(case["case_id"], f"cases[{index}].case_id")
|
|
if case_id in seen:
|
|
raise ValueError(f"duplicate case_id: {case_id}")
|
|
seen.add(case_id)
|
|
_nonempty_text(case["description"], f"cases[{index}].description")
|
|
validate_evidence_units(case["evidence_units"])
|
|
if not isinstance(case["expected"], dict):
|
|
raise ValueError(f"cases[{index}].expected must be an object")
|
|
return cases
|
|
|
|
|
|
def run_case(
|
|
case: dict[str, Any],
|
|
output_root: Path,
|
|
endpoint: str,
|
|
model: str,
|
|
timeout: int,
|
|
num_ctx: int,
|
|
num_predict: int,
|
|
) -> dict[str, Any]:
|
|
case_dir = output_root / case["case_id"]
|
|
case_dir.mkdir(parents=True, exist_ok=False)
|
|
input_payload = {
|
|
"case_id": case["case_id"],
|
|
"description": case["description"],
|
|
"evidence_units": case["evidence_units"],
|
|
}
|
|
(case_dir / "input.json").write_text(
|
|
json.dumps(input_payload, ensure_ascii=False, indent=2) + "\n",
|
|
encoding="utf-8",
|
|
)
|
|
prompt = build_prompt(case)
|
|
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
|
|
|
|
started = time.perf_counter()
|
|
try:
|
|
raw_text, metadata = call_ollama(
|
|
endpoint, model, prompt, timeout, num_ctx, num_predict
|
|
)
|
|
(case_dir / "raw_model_response.txt").write_text(
|
|
raw_text + "\n", encoding="utf-8"
|
|
)
|
|
(case_dir / "ollama_metadata.json").write_text(
|
|
json.dumps(metadata, ensure_ascii=False, indent=2) + "\n",
|
|
encoding="utf-8",
|
|
)
|
|
parsed = parse_model_json(raw_text)
|
|
(case_dir / "parsed_output.json").write_text(
|
|
json.dumps(parsed, ensure_ascii=False, indent=2) + "\n",
|
|
encoding="utf-8",
|
|
)
|
|
validated = validate_reconstruction(parsed, case["evidence_units"])
|
|
evaluation = evaluate_reconstruction(validated, case["expected"])
|
|
except requests.RequestException as exc:
|
|
failure = {
|
|
"case_id": case["case_id"],
|
|
"error_type": type(exc).__name__,
|
|
"error": str(exc),
|
|
"elapsed_seconds": round(time.perf_counter() - started, 3),
|
|
}
|
|
(case_dir / "validation_failure.json").write_text(
|
|
json.dumps(failure, ensure_ascii=False, indent=2) + "\n",
|
|
encoding="utf-8",
|
|
)
|
|
raise
|
|
except (json.JSONDecodeError, ReconstructionValidationError, ValueError) as exc:
|
|
elapsed = round(time.perf_counter() - started, 3)
|
|
failure = {
|
|
"case_id": case["case_id"],
|
|
"error_type": type(exc).__name__,
|
|
"error": str(exc),
|
|
"elapsed_seconds": elapsed,
|
|
}
|
|
(case_dir / "validation_failure.json").write_text(
|
|
json.dumps(failure, ensure_ascii=False, indent=2) + "\n",
|
|
encoding="utf-8",
|
|
)
|
|
result = {
|
|
"case_id": case["case_id"],
|
|
"description": case["description"],
|
|
"verdict": "FAIL",
|
|
"reason": f"{type(exc).__name__}: {exc}",
|
|
"passed_checks": 0,
|
|
"check_count": 0,
|
|
"critical_failures": ["schema_validation"],
|
|
"checks": [],
|
|
"elapsed_seconds": elapsed,
|
|
"subject_titles": [],
|
|
}
|
|
(case_dir / "evaluation.json").write_text(
|
|
json.dumps(result, ensure_ascii=False, indent=2) + "\n",
|
|
encoding="utf-8",
|
|
)
|
|
return result
|
|
|
|
result = {
|
|
"case_id": case["case_id"],
|
|
"description": case["description"],
|
|
**evaluation,
|
|
"elapsed_seconds": metadata["elapsed_seconds"],
|
|
"subject_titles": [item["title"] for item in validated["subjects"]],
|
|
}
|
|
(case_dir / "evaluation.json").write_text(
|
|
json.dumps(result, ensure_ascii=False, indent=2) + "\n",
|
|
encoding="utf-8",
|
|
)
|
|
return result
|
|
|
|
|
|
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
|
|
cases = load_fixture(args.fixture)
|
|
selected = set(args.case_ids or [])
|
|
if selected:
|
|
known = {case["case_id"] for case in cases}
|
|
unknown = selected - known
|
|
if unknown:
|
|
raise ValueError(f"unknown requested case IDs: {sorted(unknown)}")
|
|
cases = [case for case in cases if case["case_id"] in selected]
|
|
|
|
args.output.mkdir(parents=True, exist_ok=False)
|
|
results: list[dict[str, Any]] = []
|
|
started = time.perf_counter()
|
|
for index, case in enumerate(cases, start=1):
|
|
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
|
|
results.append(
|
|
run_case(
|
|
case,
|
|
args.output,
|
|
args.endpoint,
|
|
args.model,
|
|
args.timeout,
|
|
args.num_ctx,
|
|
args.num_predict,
|
|
)
|
|
)
|
|
summary = {
|
|
"experiment": "topic_reconstruction_v2",
|
|
"schema_version": SCHEMA_VERSION,
|
|
"model": args.model,
|
|
"temperature": 0,
|
|
"think": False,
|
|
"case_count": len(cases),
|
|
"llm_call_count": len(results),
|
|
"runtime_seconds": round(time.perf_counter() - started, 3),
|
|
"verdict_counts": {
|
|
verdict: sum(item["verdict"] == verdict for item in results)
|
|
for verdict in ("PASS", "PARTIAL", "FAIL")
|
|
},
|
|
"results": results,
|
|
}
|
|
(args.output / "summary.json").write_text(
|
|
json.dumps(summary, ensure_ascii=False, indent=2) + "\n",
|
|
encoding="utf-8",
|
|
)
|
|
return summary
|
|
|
|
|
|
def main() -> int:
|
|
args = parse_args()
|
|
try:
|
|
summary = run_experiment(args)
|
|
except (OSError, ValueError, requests.RequestException) as exc:
|
|
print(f"Error: {exc}")
|
|
return 1
|
|
print(json.dumps(summary["verdict_counts"], sort_keys=True))
|
|
print(f"Artifacts: {args.output.resolve()}")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|