Add evidence-near semantic architecture experiments
Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
This commit is contained in:
@@ -0,0 +1,721 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run an isolated Discussion Subject reconstruction experiment with Ollama."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
SCHEMA_VERSION = "experimental-discussion-subjects-v1"
|
||||
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
|
||||
DEFAULT_MODEL = "qwen3.5:9B"
|
||||
DEFAULT_TIMEOUT = 300
|
||||
DEFAULT_NUM_CTX = 16384
|
||||
DEFAULT_NUM_PREDICT = 4096
|
||||
|
||||
EVENT_TYPES = {
|
||||
"introduced_idea",
|
||||
"considered_option",
|
||||
"proposal",
|
||||
"supporting_argument",
|
||||
"objection",
|
||||
"clarification",
|
||||
"modification",
|
||||
"fact",
|
||||
"technical_finding",
|
||||
}
|
||||
OUTCOME_CERTAINTIES = {"established", "tentative", "conditional", "rejected"}
|
||||
IDENTIFIER_RE = re.compile(r"^[a-z][a-z0-9_]*$")
|
||||
|
||||
|
||||
class ReconstructionValidationError(ValueError):
|
||||
"""Raised when experimental reconstruction output violates the schema."""
|
||||
|
||||
|
||||
PROMPT_TEMPLATE = """You reconstruct discussion subjects from meeting evidence.
|
||||
|
||||
This is semantic reconstruction, not protocol writing and not flat category extraction.
|
||||
Group evidence by what participants are actually discussing. For each subject, record
|
||||
only supported discourse events and, when present, the actual outcome, resulting
|
||||
actions, and genuinely unresolved issues.
|
||||
|
||||
Important distinctions:
|
||||
- discussed is not necessarily proposed
|
||||
- proposed is not necessarily preferred or accepted
|
||||
- preferred is not accepted
|
||||
- accepted for a trial is not accepted as a final solution
|
||||
- mentioned is not an unresolved question
|
||||
- an outcome must preserve its scope, conditions, polarity, and uncertainty
|
||||
- do not infer responsibility from mention, expertise, adjacency, or likely role
|
||||
- do not invent missing stages or emit empty optional structures
|
||||
|
||||
Evidence discipline:
|
||||
- Use only the supplied evidence IDs in evidence_refs.
|
||||
- Every subject, event, outcome, action, and unresolved issue needs at least one
|
||||
evidence reference.
|
||||
- Keep statements concise; do not copy long evidence passages.
|
||||
- A subject may consist only of one introduced idea.
|
||||
|
||||
Return one JSON object with exactly:
|
||||
{{
|
||||
"schema_version": "experimental-discussion-subjects-v1",
|
||||
"subjects": [
|
||||
{{
|
||||
"subject_id": "subject_1",
|
||||
"title": "concise discussion subject",
|
||||
"evidence_refs": ["e1"],
|
||||
"development": [
|
||||
{{
|
||||
"event_id": "event_1",
|
||||
"type": "introduced_idea|considered_option|proposal|supporting_argument|objection|clarification|modification|fact|technical_finding",
|
||||
"text": "what happened in the discussion",
|
||||
"evidence_refs": ["e1"]
|
||||
}}
|
||||
],
|
||||
"outcome": {{
|
||||
"text": "only what was established",
|
||||
"scope": "explicit limit or full scope of the outcome",
|
||||
"certainty": "established|tentative|conditional|rejected",
|
||||
"evidence_refs": ["e2"]
|
||||
}},
|
||||
"actions": [
|
||||
{{
|
||||
"action_id": "action_1",
|
||||
"text": "established work only",
|
||||
"responsible": "explicitly supported name or null",
|
||||
"deadline": "explicitly supported deadline or null",
|
||||
"evidence_refs": ["e3"]
|
||||
}}
|
||||
],
|
||||
"unresolved_issues": [
|
||||
{{
|
||||
"issue_id": "issue_1",
|
||||
"text": "concrete unresolved issue",
|
||||
"evidence_refs": ["e4"]
|
||||
}}
|
||||
]
|
||||
}}
|
||||
]
|
||||
}}
|
||||
|
||||
Only subject_id, title, evidence_refs are required for each subject. Omit
|
||||
development, outcome, actions, or unresolved_issues when absent. Never emit null
|
||||
or an empty optional list/object.
|
||||
|
||||
Case ID: {case_id}
|
||||
Evidence units:
|
||||
{evidence_json}
|
||||
"""
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Run the isolated topic-reconstruction Gold experiment."
|
||||
)
|
||||
parser.add_argument("fixture", type=Path, help="Focused Gold cases JSON.")
|
||||
parser.add_argument("-o", "--output", type=Path, required=True)
|
||||
parser.add_argument("--model", default=DEFAULT_MODEL)
|
||||
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
|
||||
parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT)
|
||||
parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX)
|
||||
parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT)
|
||||
parser.add_argument(
|
||||
"--case", action="append", dest="case_ids", help="Run only this case ID."
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def _expect_exact_keys(
|
||||
value: dict[str, Any], required: set[str], optional: set[str], location: str
|
||||
) -> None:
|
||||
missing = required - value.keys()
|
||||
unknown = value.keys() - required - optional
|
||||
if missing:
|
||||
raise ReconstructionValidationError(
|
||||
f"{location} missing required keys: {sorted(missing)}"
|
||||
)
|
||||
if unknown:
|
||||
raise ReconstructionValidationError(
|
||||
f"{location} has unknown keys: {sorted(unknown)}"
|
||||
)
|
||||
|
||||
|
||||
def _nonempty_text(value: Any, location: str) -> str:
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise ReconstructionValidationError(f"{location} must be a non-empty string")
|
||||
return value.strip()
|
||||
|
||||
|
||||
def _identifier(value: Any, location: str, seen: set[str]) -> str:
|
||||
text = _nonempty_text(value, location)
|
||||
if not IDENTIFIER_RE.fullmatch(text):
|
||||
raise ReconstructionValidationError(f"{location} is not a valid identifier")
|
||||
if text in seen:
|
||||
raise ReconstructionValidationError(f"duplicate identifier: {text}")
|
||||
seen.add(text)
|
||||
return text
|
||||
|
||||
|
||||
def _nullable_text(value: Any, location: str) -> str | None:
|
||||
if value is None:
|
||||
return None
|
||||
text = _nonempty_text(value, location)
|
||||
if text.casefold() == "null":
|
||||
raise ReconstructionValidationError(
|
||||
f"{location} must use JSON null, not the string 'null'"
|
||||
)
|
||||
return text
|
||||
|
||||
|
||||
def _evidence_refs(value: Any, location: str, known: set[str]) -> list[str]:
|
||||
if not isinstance(value, list) or not value:
|
||||
raise ReconstructionValidationError(f"{location} must be a non-empty list")
|
||||
refs: list[str] = []
|
||||
for index, ref in enumerate(value):
|
||||
ref = _nonempty_text(ref, f"{location}[{index}]")
|
||||
if ref not in known:
|
||||
raise ReconstructionValidationError(
|
||||
f"{location}[{index}] references unknown evidence ID: {ref}"
|
||||
)
|
||||
if ref in refs:
|
||||
raise ReconstructionValidationError(
|
||||
f"{location} contains duplicate evidence reference: {ref}"
|
||||
)
|
||||
refs.append(ref)
|
||||
return refs
|
||||
|
||||
|
||||
def validate_evidence_units(evidence_units: Any) -> set[str]:
|
||||
if not isinstance(evidence_units, list) or not evidence_units:
|
||||
raise ReconstructionValidationError("evidence_units must be a non-empty list")
|
||||
known: set[str] = set()
|
||||
for index, unit in enumerate(evidence_units):
|
||||
location = f"evidence_units[{index}]"
|
||||
if not isinstance(unit, dict):
|
||||
raise ReconstructionValidationError(f"{location} must be an object")
|
||||
_expect_exact_keys(unit, {"evidence_id", "text"}, set(), location)
|
||||
evidence_id = _nonempty_text(unit["evidence_id"], f"{location}.evidence_id")
|
||||
if evidence_id in known:
|
||||
raise ReconstructionValidationError(
|
||||
f"duplicate input evidence identifier: {evidence_id}"
|
||||
)
|
||||
known.add(evidence_id)
|
||||
_nonempty_text(unit["text"], f"{location}.text")
|
||||
return known
|
||||
|
||||
|
||||
def validate_reconstruction(data: Any, evidence_units: Any) -> dict[str, Any]:
|
||||
known = validate_evidence_units(evidence_units)
|
||||
if not isinstance(data, dict):
|
||||
raise ReconstructionValidationError("model output must be an object")
|
||||
_expect_exact_keys(data, {"schema_version", "subjects"}, set(), "output")
|
||||
if data["schema_version"] != SCHEMA_VERSION:
|
||||
raise ReconstructionValidationError(
|
||||
f"schema_version must be {SCHEMA_VERSION!r}"
|
||||
)
|
||||
subjects = data["subjects"]
|
||||
if not isinstance(subjects, list) or not subjects:
|
||||
raise ReconstructionValidationError("subjects must be a non-empty list")
|
||||
|
||||
seen: set[str] = set()
|
||||
for subject_index, subject in enumerate(subjects):
|
||||
location = f"subjects[{subject_index}]"
|
||||
if not isinstance(subject, dict):
|
||||
raise ReconstructionValidationError(f"{location} must be an object")
|
||||
_expect_exact_keys(
|
||||
subject,
|
||||
{"subject_id", "title", "evidence_refs"},
|
||||
{"development", "outcome", "actions", "unresolved_issues"},
|
||||
location,
|
||||
)
|
||||
_identifier(subject["subject_id"], f"{location}.subject_id", seen)
|
||||
_nonempty_text(subject["title"], f"{location}.title")
|
||||
_evidence_refs(subject["evidence_refs"], f"{location}.evidence_refs", known)
|
||||
|
||||
if "development" in subject:
|
||||
events = subject["development"]
|
||||
if not isinstance(events, list) or not events:
|
||||
raise ReconstructionValidationError(
|
||||
f"{location}.development must be a non-empty list when present"
|
||||
)
|
||||
for event_index, event in enumerate(events):
|
||||
event_location = f"{location}.development[{event_index}]"
|
||||
if not isinstance(event, dict):
|
||||
raise ReconstructionValidationError(
|
||||
f"{event_location} must be an object"
|
||||
)
|
||||
_expect_exact_keys(
|
||||
event,
|
||||
{"event_id", "type", "text", "evidence_refs"},
|
||||
set(),
|
||||
event_location,
|
||||
)
|
||||
_identifier(event["event_id"], f"{event_location}.event_id", seen)
|
||||
if event["type"] not in EVENT_TYPES:
|
||||
raise ReconstructionValidationError(
|
||||
f"{event_location}.type is invalid: {event['type']!r}"
|
||||
)
|
||||
_nonempty_text(event["text"], f"{event_location}.text")
|
||||
_evidence_refs(
|
||||
event["evidence_refs"], f"{event_location}.evidence_refs", known
|
||||
)
|
||||
|
||||
if "outcome" in subject:
|
||||
outcome = subject["outcome"]
|
||||
outcome_location = f"{location}.outcome"
|
||||
if not isinstance(outcome, dict):
|
||||
raise ReconstructionValidationError(
|
||||
f"{outcome_location} must be a non-empty object when present"
|
||||
)
|
||||
_expect_exact_keys(
|
||||
outcome,
|
||||
{"text", "scope", "certainty", "evidence_refs"},
|
||||
set(),
|
||||
outcome_location,
|
||||
)
|
||||
_nonempty_text(outcome["text"], f"{outcome_location}.text")
|
||||
_nonempty_text(outcome["scope"], f"{outcome_location}.scope")
|
||||
if outcome["certainty"] not in OUTCOME_CERTAINTIES:
|
||||
raise ReconstructionValidationError(
|
||||
f"{outcome_location}.certainty is invalid: {outcome['certainty']!r}"
|
||||
)
|
||||
_evidence_refs(
|
||||
outcome["evidence_refs"], f"{outcome_location}.evidence_refs", known
|
||||
)
|
||||
|
||||
if "actions" in subject:
|
||||
actions = subject["actions"]
|
||||
if not isinstance(actions, list) or not actions:
|
||||
raise ReconstructionValidationError(
|
||||
f"{location}.actions must be a non-empty list when present"
|
||||
)
|
||||
for action_index, action in enumerate(actions):
|
||||
action_location = f"{location}.actions[{action_index}]"
|
||||
if not isinstance(action, dict):
|
||||
raise ReconstructionValidationError(
|
||||
f"{action_location} must be an object"
|
||||
)
|
||||
_expect_exact_keys(
|
||||
action,
|
||||
{"action_id", "text", "responsible", "deadline", "evidence_refs"},
|
||||
set(),
|
||||
action_location,
|
||||
)
|
||||
_identifier(action["action_id"], f"{action_location}.action_id", seen)
|
||||
_nonempty_text(action["text"], f"{action_location}.text")
|
||||
for field in ("responsible", "deadline"):
|
||||
_nullable_text(action[field], f"{action_location}.{field}")
|
||||
_evidence_refs(
|
||||
action["evidence_refs"], f"{action_location}.evidence_refs", known
|
||||
)
|
||||
|
||||
if "unresolved_issues" in subject:
|
||||
issues = subject["unresolved_issues"]
|
||||
if not isinstance(issues, list) or not issues:
|
||||
raise ReconstructionValidationError(
|
||||
f"{location}.unresolved_issues must be a non-empty list when present"
|
||||
)
|
||||
for issue_index, issue in enumerate(issues):
|
||||
issue_location = f"{location}.unresolved_issues[{issue_index}]"
|
||||
if not isinstance(issue, dict):
|
||||
raise ReconstructionValidationError(
|
||||
f"{issue_location} must be an object"
|
||||
)
|
||||
_expect_exact_keys(
|
||||
issue,
|
||||
{"issue_id", "text", "evidence_refs"},
|
||||
set(),
|
||||
issue_location,
|
||||
)
|
||||
_identifier(issue["issue_id"], f"{issue_location}.issue_id", seen)
|
||||
_nonempty_text(issue["text"], f"{issue_location}.text")
|
||||
_evidence_refs(
|
||||
issue["evidence_refs"], f"{issue_location}.evidence_refs", known
|
||||
)
|
||||
|
||||
return data
|
||||
|
||||
|
||||
def build_prompt(case: dict[str, Any]) -> str:
|
||||
evidence_units = case["evidence_units"]
|
||||
validate_evidence_units(evidence_units)
|
||||
return PROMPT_TEMPLATE.format(
|
||||
case_id=case["case_id"],
|
||||
evidence_json=json.dumps(evidence_units, ensure_ascii=False, indent=2),
|
||||
)
|
||||
|
||||
|
||||
def parse_model_json(raw_text: str) -> dict[str, Any]:
|
||||
data = json.loads(raw_text)
|
||||
if not isinstance(data, dict):
|
||||
raise ReconstructionValidationError("model response JSON must be an object")
|
||||
return data
|
||||
|
||||
|
||||
def build_ollama_payload(
|
||||
model: str, prompt: str, num_ctx: int, num_predict: int
|
||||
) -> dict[str, Any]:
|
||||
return {
|
||||
"model": model,
|
||||
"prompt": prompt,
|
||||
"think": False,
|
||||
"stream": False,
|
||||
"format": "json",
|
||||
"options": {
|
||||
"temperature": 0,
|
||||
"num_ctx": num_ctx,
|
||||
"num_predict": num_predict,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def call_ollama(
|
||||
endpoint: str,
|
||||
model: str,
|
||||
prompt: str,
|
||||
timeout: int,
|
||||
num_ctx: int,
|
||||
num_predict: int,
|
||||
) -> tuple[str, dict[str, Any]]:
|
||||
payload = build_ollama_payload(model, prompt, num_ctx, num_predict)
|
||||
started = time.perf_counter()
|
||||
response = requests.post(endpoint, json=payload, timeout=timeout)
|
||||
elapsed = time.perf_counter() - started
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
if not isinstance(data, dict):
|
||||
raise ValueError("Ollama response must be a JSON object")
|
||||
raw_text = data.get("response")
|
||||
if not isinstance(raw_text, str) or not raw_text.strip():
|
||||
raise ValueError("Ollama returned no usable response text")
|
||||
metadata = {
|
||||
"model": data.get("model", model),
|
||||
"elapsed_seconds": round(elapsed, 3),
|
||||
"total_duration_ns": data.get("total_duration"),
|
||||
"load_duration_ns": data.get("load_duration"),
|
||||
"prompt_eval_count": data.get("prompt_eval_count"),
|
||||
"prompt_eval_duration_ns": data.get("prompt_eval_duration"),
|
||||
"eval_count": data.get("eval_count"),
|
||||
"eval_duration_ns": data.get("eval_duration"),
|
||||
"configuration": {
|
||||
"temperature": 0,
|
||||
"think": False,
|
||||
"num_ctx": num_ctx,
|
||||
"num_predict": num_predict,
|
||||
},
|
||||
}
|
||||
return raw_text.strip(), metadata
|
||||
|
||||
|
||||
def _all_text(subjects: list[dict[str, Any]]) -> str:
|
||||
parts: list[str] = []
|
||||
for subject in subjects:
|
||||
parts.append(subject["title"])
|
||||
for event in subject.get("development", []):
|
||||
parts.append(event["text"])
|
||||
outcome = subject.get("outcome")
|
||||
if outcome:
|
||||
parts.extend((outcome["text"], outcome["scope"]))
|
||||
for action in subject.get("actions", []):
|
||||
parts.append(action["text"])
|
||||
for issue in subject.get("unresolved_issues", []):
|
||||
parts.append(issue["text"])
|
||||
return " ".join(parts).casefold()
|
||||
|
||||
|
||||
def _contains_any(text: str, terms: list[str]) -> bool:
|
||||
return any(term.casefold() in text for term in terms)
|
||||
|
||||
|
||||
def evaluate_reconstruction(
|
||||
reconstruction: dict[str, Any], expected: dict[str, Any]
|
||||
) -> dict[str, Any]:
|
||||
subjects = reconstruction["subjects"]
|
||||
combined = _all_text(subjects)
|
||||
events = [event for subject in subjects for event in subject.get("development", [])]
|
||||
outcomes = [subject["outcome"] for subject in subjects if "outcome" in subject]
|
||||
actions = [action for subject in subjects for action in subject.get("actions", [])]
|
||||
issues = [issue for subject in subjects for issue in subject.get("unresolved_issues", [])]
|
||||
checks: list[dict[str, Any]] = []
|
||||
|
||||
def add(name: str, passed: bool, critical: bool = False) -> None:
|
||||
checks.append({"name": name, "passed": passed, "critical": critical})
|
||||
|
||||
add("subject_count", len(subjects) == expected.get("subject_count", 1))
|
||||
add("subject_identity", _contains_any(combined, expected["subject_terms"]))
|
||||
|
||||
event_types = {event["type"] for event in events}
|
||||
for event_type in expected.get("required_event_types", []):
|
||||
add(f"event_type:{event_type}", event_type in event_types)
|
||||
|
||||
expected_outcome = expected.get("outcome", {})
|
||||
outcome_required = expected_outcome.get("required", False)
|
||||
add(
|
||||
"outcome_presence",
|
||||
bool(outcomes) is outcome_required,
|
||||
critical=not outcome_required and bool(outcomes),
|
||||
)
|
||||
if outcome_required and outcomes:
|
||||
outcome_text = " ".join(
|
||||
f"{item['text']} {item['scope']}" for item in outcomes
|
||||
).casefold()
|
||||
add("outcome_meaning", _contains_any(outcome_text, expected_outcome["terms"]))
|
||||
add(
|
||||
"outcome_scope",
|
||||
_contains_any(outcome_text, expected_outcome.get("scope_terms", [])),
|
||||
critical=True,
|
||||
)
|
||||
add(
|
||||
"outcome_certainty",
|
||||
any(
|
||||
item["certainty"] in expected_outcome.get("certainties", [])
|
||||
for item in outcomes
|
||||
),
|
||||
)
|
||||
|
||||
expected_actions = expected.get("actions", {})
|
||||
minimum_actions = expected_actions.get("minimum", 0)
|
||||
add(
|
||||
"action_count",
|
||||
len(actions) >= minimum_actions if minimum_actions else not actions,
|
||||
critical=minimum_actions == 0 and bool(actions),
|
||||
)
|
||||
if minimum_actions and actions:
|
||||
action_text = " ".join(item["text"] for item in actions).casefold()
|
||||
add("action_meaning", _contains_any(action_text, expected_actions["terms"]))
|
||||
if "responsible" in expected_actions:
|
||||
add(
|
||||
"action_responsibility",
|
||||
any(
|
||||
item["responsible"] == expected_actions["responsible"]
|
||||
for item in actions
|
||||
),
|
||||
critical=True,
|
||||
)
|
||||
|
||||
expected_issues = expected.get("unresolved", {})
|
||||
minimum_issues = expected_issues.get("minimum", 0)
|
||||
add(
|
||||
"unresolved_count",
|
||||
len(issues) >= minimum_issues if minimum_issues else not issues,
|
||||
critical=minimum_issues == 0 and bool(issues),
|
||||
)
|
||||
if minimum_issues and issues:
|
||||
issue_text = " ".join(item["text"] for item in issues).casefold()
|
||||
add("unresolved_meaning", _contains_any(issue_text, expected_issues["terms"]))
|
||||
|
||||
passed = sum(check["passed"] for check in checks)
|
||||
critical_failures = [
|
||||
check["name"] for check in checks if check["critical"] and not check["passed"]
|
||||
]
|
||||
ratio = passed / len(checks)
|
||||
if ratio == 1:
|
||||
verdict = "PASS"
|
||||
elif ratio >= 0.6 and not critical_failures:
|
||||
verdict = "PARTIAL"
|
||||
else:
|
||||
verdict = "FAIL"
|
||||
failed = [check["name"] for check in checks if not check["passed"]]
|
||||
reason = "All semantic checks passed." if not failed else "Failed: " + ", ".join(failed)
|
||||
return {
|
||||
"verdict": verdict,
|
||||
"reason": reason,
|
||||
"passed_checks": passed,
|
||||
"check_count": len(checks),
|
||||
"critical_failures": critical_failures,
|
||||
"checks": checks,
|
||||
}
|
||||
|
||||
|
||||
def load_fixture(path: Path) -> list[dict[str, Any]]:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
if not isinstance(data, dict) or set(data) != {"cases"}:
|
||||
raise ValueError("fixture must contain exactly one 'cases' list")
|
||||
cases = data["cases"]
|
||||
if not isinstance(cases, list) or not cases:
|
||||
raise ValueError("fixture cases must be a non-empty list")
|
||||
seen: set[str] = set()
|
||||
for index, case in enumerate(cases):
|
||||
if not isinstance(case, dict):
|
||||
raise ValueError(f"cases[{index}] must be an object")
|
||||
required = {"case_id", "description", "evidence_units", "expected"}
|
||||
if set(case) != required:
|
||||
raise ValueError(f"cases[{index}] must contain exactly {sorted(required)}")
|
||||
case_id = _nonempty_text(case["case_id"], f"cases[{index}].case_id")
|
||||
if case_id in seen:
|
||||
raise ValueError(f"duplicate case_id: {case_id}")
|
||||
seen.add(case_id)
|
||||
_nonempty_text(case["description"], f"cases[{index}].description")
|
||||
validate_evidence_units(case["evidence_units"])
|
||||
if not isinstance(case["expected"], dict):
|
||||
raise ValueError(f"cases[{index}].expected must be an object")
|
||||
return cases
|
||||
|
||||
|
||||
def run_case(
|
||||
case: dict[str, Any],
|
||||
output_root: Path,
|
||||
endpoint: str,
|
||||
model: str,
|
||||
timeout: int,
|
||||
num_ctx: int,
|
||||
num_predict: int,
|
||||
) -> dict[str, Any]:
|
||||
case_dir = output_root / case["case_id"]
|
||||
case_dir.mkdir(parents=True, exist_ok=False)
|
||||
input_payload = {
|
||||
"case_id": case["case_id"],
|
||||
"description": case["description"],
|
||||
"evidence_units": case["evidence_units"],
|
||||
}
|
||||
(case_dir / "input.json").write_text(
|
||||
json.dumps(input_payload, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
prompt = build_prompt(case)
|
||||
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
|
||||
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
raw_text, metadata = call_ollama(
|
||||
endpoint, model, prompt, timeout, num_ctx, num_predict
|
||||
)
|
||||
(case_dir / "raw_model_response.txt").write_text(
|
||||
raw_text + "\n", encoding="utf-8"
|
||||
)
|
||||
(case_dir / "ollama_metadata.json").write_text(
|
||||
json.dumps(metadata, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
parsed = parse_model_json(raw_text)
|
||||
(case_dir / "parsed_output.json").write_text(
|
||||
json.dumps(parsed, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
validated = validate_reconstruction(parsed, case["evidence_units"])
|
||||
evaluation = evaluate_reconstruction(validated, case["expected"])
|
||||
except requests.RequestException as exc:
|
||||
failure = {
|
||||
"case_id": case["case_id"],
|
||||
"error_type": type(exc).__name__,
|
||||
"error": str(exc),
|
||||
"elapsed_seconds": round(time.perf_counter() - started, 3),
|
||||
}
|
||||
(case_dir / "validation_failure.json").write_text(
|
||||
json.dumps(failure, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
raise
|
||||
except (json.JSONDecodeError, ReconstructionValidationError, ValueError) as exc:
|
||||
elapsed = round(time.perf_counter() - started, 3)
|
||||
failure = {
|
||||
"case_id": case["case_id"],
|
||||
"error_type": type(exc).__name__,
|
||||
"error": str(exc),
|
||||
"elapsed_seconds": elapsed,
|
||||
}
|
||||
(case_dir / "validation_failure.json").write_text(
|
||||
json.dumps(failure, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
result = {
|
||||
"case_id": case["case_id"],
|
||||
"description": case["description"],
|
||||
"verdict": "FAIL",
|
||||
"reason": f"{type(exc).__name__}: {exc}",
|
||||
"passed_checks": 0,
|
||||
"check_count": 0,
|
||||
"critical_failures": ["schema_validation"],
|
||||
"checks": [],
|
||||
"elapsed_seconds": elapsed,
|
||||
"subject_titles": [],
|
||||
}
|
||||
(case_dir / "evaluation.json").write_text(
|
||||
json.dumps(result, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
return result
|
||||
|
||||
result = {
|
||||
"case_id": case["case_id"],
|
||||
"description": case["description"],
|
||||
**evaluation,
|
||||
"elapsed_seconds": metadata["elapsed_seconds"],
|
||||
"subject_titles": [item["title"] for item in validated["subjects"]],
|
||||
}
|
||||
(case_dir / "evaluation.json").write_text(
|
||||
json.dumps(result, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
|
||||
cases = load_fixture(args.fixture)
|
||||
selected = set(args.case_ids or [])
|
||||
if selected:
|
||||
known = {case["case_id"] for case in cases}
|
||||
unknown = selected - known
|
||||
if unknown:
|
||||
raise ValueError(f"unknown requested case IDs: {sorted(unknown)}")
|
||||
cases = [case for case in cases if case["case_id"] in selected]
|
||||
|
||||
args.output.mkdir(parents=True, exist_ok=False)
|
||||
results: list[dict[str, Any]] = []
|
||||
started = time.perf_counter()
|
||||
for index, case in enumerate(cases, start=1):
|
||||
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
|
||||
results.append(
|
||||
run_case(
|
||||
case,
|
||||
args.output,
|
||||
args.endpoint,
|
||||
args.model,
|
||||
args.timeout,
|
||||
args.num_ctx,
|
||||
args.num_predict,
|
||||
)
|
||||
)
|
||||
summary = {
|
||||
"experiment": "topic_reconstruction_v2",
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"model": args.model,
|
||||
"temperature": 0,
|
||||
"think": False,
|
||||
"case_count": len(cases),
|
||||
"llm_call_count": len(results),
|
||||
"runtime_seconds": round(time.perf_counter() - started, 3),
|
||||
"verdict_counts": {
|
||||
verdict: sum(item["verdict"] == verdict for item in results)
|
||||
for verdict in ("PASS", "PARTIAL", "FAIL")
|
||||
},
|
||||
"results": results,
|
||||
}
|
||||
(args.output / "summary.json").write_text(
|
||||
json.dumps(summary, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
return summary
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
try:
|
||||
summary = run_experiment(args)
|
||||
except (OSError, ValueError, requests.RequestException) as exc:
|
||||
print(f"Error: {exc}")
|
||||
return 1
|
||||
print(json.dumps(summary["verdict_counts"], sort_keys=True))
|
||||
print(f"Artifacts: {args.output.resolve()}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user