Add evidence-near semantic architecture experiments
Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
This commit is contained in:
@@ -0,0 +1,640 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run semantic synthesis with subject detection and evidence assignment fixed."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
SCHEMA_VERSION = "experimental-semantic-synthesis-v1"
|
||||
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
|
||||
DEFAULT_MODEL = "qwen3.5:9B"
|
||||
DEFAULT_TIMEOUT = 300
|
||||
DEFAULT_NUM_CTX = 8192
|
||||
DEFAULT_NUM_PREDICT = 2048
|
||||
|
||||
EVENT_TYPES = {
|
||||
"idea",
|
||||
"option",
|
||||
"proposal",
|
||||
"objection",
|
||||
"supporting_argument",
|
||||
"clarification",
|
||||
"rejection",
|
||||
"scoped_acceptance",
|
||||
"fact",
|
||||
"technical_finding",
|
||||
}
|
||||
OUTCOME_STATUSES = {"established", "rejected", "scoped_acceptance", "tentative"}
|
||||
|
||||
|
||||
class SynthesisValidationError(ValueError):
|
||||
"""Raised when isolated semantic synthesis output is structurally invalid."""
|
||||
|
||||
|
||||
PROMPT_TEMPLATE = """You perform semantic synthesis for one already known discussion subject.
|
||||
|
||||
The subject boundary and evidence assignment are fixed and complete. Do not discover,
|
||||
split, merge, rename, or omit the subject. Do not assign evidence to another subject.
|
||||
Interpret only what the supplied evidence semantically establishes.
|
||||
|
||||
Semantic distinctions:
|
||||
- idea: mentioned possibility without stronger commitment
|
||||
- option: alternative considered without commitment
|
||||
- proposal: suggested course of action not yet established as work
|
||||
- objection: argument or concern against something; not automatically unresolved
|
||||
- rejection: an alternative is explicitly rejected
|
||||
- scoped_acceptance: accepted only for the stated test, trial, condition, or scope
|
||||
- proposal is not an action
|
||||
- no decision is not a tentative decision
|
||||
- mention is not an unresolved issue
|
||||
- an action requires explicit assignment, acceptance, commitment, or established work
|
||||
- an unresolved issue requires a concrete need explicitly left unresolved
|
||||
|
||||
Preserve explicit rejection, explicit accepted work, explicit unresolved questions,
|
||||
and all limits on an outcome. Never generalize trial acceptance into final acceptance.
|
||||
Use only supplied evidence IDs. Keep concise semantic text in the evidence language.
|
||||
|
||||
Return exactly one JSON object. Always include these fields:
|
||||
{{
|
||||
"schema_version": "experimental-semantic-synthesis-v1",
|
||||
"subject_id": "copy the supplied subject_id exactly",
|
||||
"subject": "copy the supplied subject exactly",
|
||||
"events": [
|
||||
{{
|
||||
"type": "idea|option|proposal|objection|supporting_argument|clarification|rejection|scoped_acceptance|fact|technical_finding",
|
||||
"text": "supported semantic event",
|
||||
"evidence_ids": ["e1"]
|
||||
}}
|
||||
],
|
||||
"actions": [
|
||||
{{
|
||||
"text": "established action",
|
||||
"responsible": null,
|
||||
"due": null,
|
||||
"evidence_ids": ["e2"]
|
||||
}}
|
||||
],
|
||||
"unresolved_issues": [
|
||||
{{
|
||||
"text": "explicitly unresolved issue",
|
||||
"evidence_ids": ["e3"]
|
||||
}}
|
||||
]
|
||||
}}
|
||||
|
||||
The three arrays are structurally required; use [] when none exist.
|
||||
Add "outcome" only when an outcome was actually established:
|
||||
{{
|
||||
"status": "established|rejected|scoped_acceptance|tentative",
|
||||
"text": "what was actually established",
|
||||
"scope": "the exact scope, condition, or limit",
|
||||
"evidence_ids": ["e2"]
|
||||
}}
|
||||
Omit outcome completely when there is none. Never use null for outcome. Never use the
|
||||
string "null"; use JSON null only for unknown responsible or due values.
|
||||
|
||||
Fixed Gold input:
|
||||
{input_json}
|
||||
"""
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Run the isolated semantic-synthesis Gold experiment."
|
||||
)
|
||||
parser.add_argument("fixture", type=Path, help="Fixed-subject Gold bundle JSON.")
|
||||
parser.add_argument("-o", "--output", type=Path, required=True)
|
||||
parser.add_argument("--model", default=DEFAULT_MODEL)
|
||||
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
|
||||
parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT)
|
||||
parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX)
|
||||
parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def _exact_keys(
|
||||
value: dict[str, Any], required: set[str], optional: set[str], location: str
|
||||
) -> None:
|
||||
missing = required - value.keys()
|
||||
unknown = value.keys() - required - optional
|
||||
if missing:
|
||||
raise SynthesisValidationError(
|
||||
f"{location} missing required keys: {sorted(missing)}"
|
||||
)
|
||||
if unknown:
|
||||
raise SynthesisValidationError(
|
||||
f"{location} has unknown keys: {sorted(unknown)}"
|
||||
)
|
||||
|
||||
|
||||
def _text(value: Any, location: str) -> str:
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise SynthesisValidationError(f"{location} must be a non-empty string")
|
||||
return value.strip()
|
||||
|
||||
|
||||
def validate_bundle(case: Any) -> dict[str, Any]:
|
||||
if not isinstance(case, dict):
|
||||
raise SynthesisValidationError("case must be an object")
|
||||
_exact_keys(
|
||||
case,
|
||||
{
|
||||
"case_id",
|
||||
"description",
|
||||
"subject_id",
|
||||
"subject",
|
||||
"evidence",
|
||||
"allowed_responsible",
|
||||
"expected",
|
||||
},
|
||||
set(),
|
||||
"case",
|
||||
)
|
||||
_text(case["case_id"], "case.case_id")
|
||||
_text(case["description"], "case.description")
|
||||
_text(case["subject_id"], "case.subject_id")
|
||||
_text(case["subject"], "case.subject")
|
||||
evidence = case["evidence"]
|
||||
if not isinstance(evidence, list) or not evidence:
|
||||
raise SynthesisValidationError("case.evidence must be a non-empty list")
|
||||
seen: set[str] = set()
|
||||
for index, item in enumerate(evidence):
|
||||
location = f"case.evidence[{index}]"
|
||||
if not isinstance(item, dict):
|
||||
raise SynthesisValidationError(f"{location} must be an object")
|
||||
_exact_keys(item, {"evidence_id", "text"}, set(), location)
|
||||
evidence_id = _text(item["evidence_id"], f"{location}.evidence_id")
|
||||
if evidence_id in seen:
|
||||
raise SynthesisValidationError(f"duplicate evidence ID: {evidence_id}")
|
||||
seen.add(evidence_id)
|
||||
_text(item["text"], f"{location}.text")
|
||||
allowed = case["allowed_responsible"]
|
||||
if not isinstance(allowed, list) or any(
|
||||
not isinstance(value, str) or not value.strip() for value in allowed
|
||||
):
|
||||
raise SynthesisValidationError(
|
||||
"case.allowed_responsible must be a list of non-empty strings"
|
||||
)
|
||||
if len(set(allowed)) != len(allowed):
|
||||
raise SynthesisValidationError("case.allowed_responsible contains duplicates")
|
||||
if not isinstance(case["expected"], dict):
|
||||
raise SynthesisValidationError("case.expected must be an object")
|
||||
return case
|
||||
|
||||
|
||||
def _evidence_ids(value: Any, location: str, known: set[str]) -> list[str]:
|
||||
if not isinstance(value, list) or not value:
|
||||
raise SynthesisValidationError(f"{location} must be a non-empty list")
|
||||
result: list[str] = []
|
||||
for index, evidence_id in enumerate(value):
|
||||
evidence_id = _text(evidence_id, f"{location}[{index}]")
|
||||
if evidence_id not in known:
|
||||
raise SynthesisValidationError(
|
||||
f"{location}[{index}] references unknown evidence ID: {evidence_id}"
|
||||
)
|
||||
if evidence_id in result:
|
||||
raise SynthesisValidationError(
|
||||
f"{location} contains duplicate evidence ID: {evidence_id}"
|
||||
)
|
||||
result.append(evidence_id)
|
||||
return result
|
||||
|
||||
|
||||
def _nullable_text(value: Any, location: str) -> str | None:
|
||||
if value is None:
|
||||
return None
|
||||
result = _text(value, location)
|
||||
if result.casefold() == "null":
|
||||
raise SynthesisValidationError(
|
||||
f"{location} must use JSON null, not the string 'null'"
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
def validate_synthesis(data: Any, case: dict[str, Any]) -> dict[str, Any]:
|
||||
validate_bundle(case)
|
||||
if not isinstance(data, dict):
|
||||
raise SynthesisValidationError("output must be an object")
|
||||
_exact_keys(
|
||||
data,
|
||||
{
|
||||
"schema_version",
|
||||
"subject_id",
|
||||
"subject",
|
||||
"events",
|
||||
"actions",
|
||||
"unresolved_issues",
|
||||
},
|
||||
{"outcome"},
|
||||
"output",
|
||||
)
|
||||
if data["schema_version"] != SCHEMA_VERSION:
|
||||
raise SynthesisValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
|
||||
if data["subject_id"] != case["subject_id"]:
|
||||
raise SynthesisValidationError("model changed fixed subject_id")
|
||||
if data["subject"] != case["subject"]:
|
||||
raise SynthesisValidationError("model changed fixed subject")
|
||||
|
||||
known = {item["evidence_id"] for item in case["evidence"]}
|
||||
events = data["events"]
|
||||
if not isinstance(events, list):
|
||||
raise SynthesisValidationError("output.events must be an array")
|
||||
for index, event in enumerate(events):
|
||||
location = f"output.events[{index}]"
|
||||
if not isinstance(event, dict):
|
||||
raise SynthesisValidationError(f"{location} must be an object")
|
||||
_exact_keys(event, {"type", "text", "evidence_ids"}, set(), location)
|
||||
if event["type"] not in EVENT_TYPES:
|
||||
raise SynthesisValidationError(f"{location}.type is invalid")
|
||||
_text(event["text"], f"{location}.text")
|
||||
_evidence_ids(event["evidence_ids"], f"{location}.evidence_ids", known)
|
||||
|
||||
if "outcome" in data:
|
||||
outcome = data["outcome"]
|
||||
if not isinstance(outcome, dict):
|
||||
raise SynthesisValidationError(
|
||||
"output.outcome must be an object when present; omit it when absent"
|
||||
)
|
||||
_exact_keys(
|
||||
outcome, {"status", "text", "scope", "evidence_ids"}, set(), "output.outcome"
|
||||
)
|
||||
if outcome["status"] not in OUTCOME_STATUSES:
|
||||
raise SynthesisValidationError("output.outcome.status is invalid")
|
||||
_text(outcome["text"], "output.outcome.text")
|
||||
_text(outcome["scope"], "output.outcome.scope")
|
||||
_evidence_ids(outcome["evidence_ids"], "output.outcome.evidence_ids", known)
|
||||
|
||||
actions = data["actions"]
|
||||
if not isinstance(actions, list):
|
||||
raise SynthesisValidationError("output.actions must be an array")
|
||||
allowed = set(case["allowed_responsible"])
|
||||
for index, action in enumerate(actions):
|
||||
location = f"output.actions[{index}]"
|
||||
if not isinstance(action, dict):
|
||||
raise SynthesisValidationError(f"{location} must be an object")
|
||||
_exact_keys(
|
||||
action,
|
||||
{"text", "responsible", "due", "evidence_ids"},
|
||||
set(),
|
||||
location,
|
||||
)
|
||||
_text(action["text"], f"{location}.text")
|
||||
responsible = _nullable_text(action["responsible"], f"{location}.responsible")
|
||||
if responsible is not None and responsible not in allowed:
|
||||
raise SynthesisValidationError(
|
||||
f"{location}.responsible is not allowed: {responsible}"
|
||||
)
|
||||
_nullable_text(action["due"], f"{location}.due")
|
||||
_evidence_ids(action["evidence_ids"], f"{location}.evidence_ids", known)
|
||||
|
||||
issues = data["unresolved_issues"]
|
||||
if not isinstance(issues, list):
|
||||
raise SynthesisValidationError("output.unresolved_issues must be an array")
|
||||
for index, issue in enumerate(issues):
|
||||
location = f"output.unresolved_issues[{index}]"
|
||||
if not isinstance(issue, dict):
|
||||
raise SynthesisValidationError(f"{location} must be an object")
|
||||
_exact_keys(issue, {"text", "evidence_ids"}, set(), location)
|
||||
_text(issue["text"], f"{location}.text")
|
||||
_evidence_ids(issue["evidence_ids"], f"{location}.evidence_ids", known)
|
||||
return data
|
||||
|
||||
|
||||
def build_prompt(case: dict[str, Any]) -> str:
|
||||
validate_bundle(case)
|
||||
model_input = {
|
||||
"subject_id": case["subject_id"],
|
||||
"subject": case["subject"],
|
||||
"evidence": case["evidence"],
|
||||
}
|
||||
return PROMPT_TEMPLATE.format(
|
||||
input_json=json.dumps(model_input, ensure_ascii=False, indent=2)
|
||||
)
|
||||
|
||||
|
||||
def parse_model_json(raw_text: str) -> dict[str, Any]:
|
||||
data = json.loads(raw_text)
|
||||
if not isinstance(data, dict):
|
||||
raise SynthesisValidationError("model response JSON must be an object")
|
||||
return data
|
||||
|
||||
|
||||
def build_ollama_payload(
|
||||
model: str, prompt: str, num_ctx: int, num_predict: int
|
||||
) -> dict[str, Any]:
|
||||
return {
|
||||
"model": model,
|
||||
"prompt": prompt,
|
||||
"think": False,
|
||||
"stream": False,
|
||||
"format": "json",
|
||||
"options": {
|
||||
"temperature": 0,
|
||||
"num_ctx": num_ctx,
|
||||
"num_predict": num_predict,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def call_ollama(
|
||||
endpoint: str,
|
||||
model: str,
|
||||
prompt: str,
|
||||
timeout: int,
|
||||
num_ctx: int,
|
||||
num_predict: int,
|
||||
) -> tuple[str, dict[str, Any]]:
|
||||
payload = build_ollama_payload(model, prompt, num_ctx, num_predict)
|
||||
started = time.perf_counter()
|
||||
response = requests.post(endpoint, json=payload, timeout=timeout)
|
||||
elapsed = time.perf_counter() - started
|
||||
response.raise_for_status()
|
||||
body = response.json()
|
||||
if not isinstance(body, dict):
|
||||
raise ValueError("Ollama response must be an object")
|
||||
raw_text = body.get("response")
|
||||
if not isinstance(raw_text, str) or not raw_text.strip():
|
||||
raise ValueError("Ollama returned no usable response text")
|
||||
metadata = {
|
||||
"model": body.get("model", model),
|
||||
"elapsed_seconds": round(elapsed, 3),
|
||||
"total_duration_ns": body.get("total_duration"),
|
||||
"load_duration_ns": body.get("load_duration"),
|
||||
"prompt_eval_count": body.get("prompt_eval_count"),
|
||||
"prompt_eval_duration_ns": body.get("prompt_eval_duration"),
|
||||
"eval_count": body.get("eval_count"),
|
||||
"eval_duration_ns": body.get("eval_duration"),
|
||||
"configuration": {
|
||||
"temperature": 0,
|
||||
"think": False,
|
||||
"num_ctx": num_ctx,
|
||||
"num_predict": num_predict,
|
||||
},
|
||||
}
|
||||
return raw_text.strip(), metadata
|
||||
|
||||
|
||||
def _contains(text: str, terms: list[str]) -> bool:
|
||||
folded = text.casefold()
|
||||
return any(term.casefold() in folded for term in terms)
|
||||
|
||||
|
||||
def _refs_cover(items: list[dict[str, Any]], expected: list[str]) -> bool:
|
||||
actual = {
|
||||
evidence_id
|
||||
for item in items
|
||||
for evidence_id in item.get("evidence_ids", [])
|
||||
}
|
||||
return set(expected).issubset(actual)
|
||||
|
||||
|
||||
def evaluate_synthesis(data: dict[str, Any], expected: dict[str, Any]) -> dict[str, Any]:
|
||||
checks: list[dict[str, Any]] = []
|
||||
|
||||
def add(name: str, passed: bool, critical: bool = False) -> None:
|
||||
checks.append({"name": name, "passed": passed, "critical": critical})
|
||||
|
||||
events = data["events"]
|
||||
event_types = [item["type"] for item in events]
|
||||
for event_type, minimum in expected.get("event_type_minimums", {}).items():
|
||||
add(f"event:{event_type}", event_types.count(event_type) >= minimum)
|
||||
allowed_types = set(expected.get("allowed_event_types", EVENT_TYPES))
|
||||
add("no_unexpected_event_types", set(event_types).issubset(allowed_types))
|
||||
add(
|
||||
"event_evidence",
|
||||
_refs_cover(events, expected.get("event_evidence_ids", [])),
|
||||
)
|
||||
|
||||
outcome_expected = expected["outcome"]
|
||||
outcome = data.get("outcome")
|
||||
add(
|
||||
"outcome_presence",
|
||||
(outcome is not None) == outcome_expected["required"],
|
||||
critical=True,
|
||||
)
|
||||
if outcome_expected["required"] and outcome is not None:
|
||||
add("outcome_status", outcome["status"] in outcome_expected["statuses"])
|
||||
combined = f"{outcome['text']} {outcome['scope']}"
|
||||
add("outcome_meaning", _contains(combined, outcome_expected["terms"]))
|
||||
add(
|
||||
"outcome_scope",
|
||||
_contains(combined, outcome_expected["scope_terms"]),
|
||||
critical=True,
|
||||
)
|
||||
add(
|
||||
"outcome_evidence",
|
||||
set(outcome_expected["evidence_ids"]).issubset(outcome["evidence_ids"]),
|
||||
critical=True,
|
||||
)
|
||||
|
||||
actions = data["actions"]
|
||||
expected_actions = expected["actions"]
|
||||
add(
|
||||
"action_count",
|
||||
len(actions) == expected_actions["count"],
|
||||
critical=True,
|
||||
)
|
||||
if expected_actions["count"] and actions:
|
||||
action_text = " ".join(item["text"] for item in actions)
|
||||
add("action_meaning", _contains(action_text, expected_actions["terms"]))
|
||||
if "responsible" in expected_actions:
|
||||
add(
|
||||
"action_responsible",
|
||||
any(item["responsible"] == expected_actions["responsible"] for item in actions),
|
||||
critical=True,
|
||||
)
|
||||
if expected_actions.get("due_terms"):
|
||||
due_text = " ".join(str(item["due"] or "") for item in actions)
|
||||
add("action_due", _contains(due_text, expected_actions["due_terms"]))
|
||||
add(
|
||||
"action_evidence",
|
||||
_refs_cover(actions, expected_actions["evidence_ids"]),
|
||||
critical=True,
|
||||
)
|
||||
|
||||
issues = data["unresolved_issues"]
|
||||
expected_issues = expected["unresolved_issues"]
|
||||
add(
|
||||
"unresolved_count",
|
||||
len(issues) == expected_issues["count"],
|
||||
critical=True,
|
||||
)
|
||||
if expected_issues["count"] and issues:
|
||||
issue_text = " ".join(item["text"] for item in issues)
|
||||
add("unresolved_meaning", _contains(issue_text, expected_issues["terms"]))
|
||||
add(
|
||||
"unresolved_evidence",
|
||||
_refs_cover(issues, expected_issues["evidence_ids"]),
|
||||
critical=True,
|
||||
)
|
||||
|
||||
passed = sum(item["passed"] for item in checks)
|
||||
critical_failures = [
|
||||
item["name"] for item in checks if item["critical"] and not item["passed"]
|
||||
]
|
||||
ratio = passed / len(checks)
|
||||
if ratio == 1:
|
||||
verdict = "PASS"
|
||||
elif ratio >= 0.7 and not critical_failures:
|
||||
verdict = "PARTIAL"
|
||||
else:
|
||||
verdict = "FAIL"
|
||||
failed = [item["name"] for item in checks if not item["passed"]]
|
||||
return {
|
||||
"verdict": verdict,
|
||||
"reason": "All semantic checks passed." if not failed else "Failed: " + ", ".join(failed),
|
||||
"passed_checks": passed,
|
||||
"check_count": len(checks),
|
||||
"critical_failures": critical_failures,
|
||||
"checks": checks,
|
||||
}
|
||||
|
||||
|
||||
def load_fixture(path: Path) -> list[dict[str, Any]]:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
if not isinstance(data, dict) or set(data) != {"cases"}:
|
||||
raise SynthesisValidationError("fixture must contain exactly a cases list")
|
||||
cases = data["cases"]
|
||||
if not isinstance(cases, list) or not cases:
|
||||
raise SynthesisValidationError("fixture cases must be a non-empty list")
|
||||
seen: set[str] = set()
|
||||
for case in cases:
|
||||
validate_bundle(case)
|
||||
if case["case_id"] in seen:
|
||||
raise SynthesisValidationError(f"duplicate case ID: {case['case_id']}")
|
||||
seen.add(case["case_id"])
|
||||
return cases
|
||||
|
||||
|
||||
def run_case(
|
||||
case: dict[str, Any],
|
||||
output_root: Path,
|
||||
endpoint: str,
|
||||
model: str,
|
||||
timeout: int,
|
||||
num_ctx: int,
|
||||
num_predict: int,
|
||||
) -> dict[str, Any]:
|
||||
case_dir = output_root / case["case_id"]
|
||||
case_dir.mkdir(parents=True, exist_ok=False)
|
||||
gold_input = {
|
||||
"case_id": case["case_id"],
|
||||
"description": case["description"],
|
||||
"subject_id": case["subject_id"],
|
||||
"subject": case["subject"],
|
||||
"evidence": case["evidence"],
|
||||
}
|
||||
(case_dir / "gold_input.json").write_text(
|
||||
json.dumps(gold_input, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
prompt = build_prompt(case)
|
||||
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
raw_text, metadata = call_ollama(
|
||||
endpoint, model, prompt, timeout, num_ctx, num_predict
|
||||
)
|
||||
(case_dir / "raw_model_response.txt").write_text(raw_text + "\n", encoding="utf-8")
|
||||
(case_dir / "ollama_metadata.json").write_text(
|
||||
json.dumps(metadata, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
parsed = parse_model_json(raw_text)
|
||||
(case_dir / "parsed_response.json").write_text(
|
||||
json.dumps(parsed, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
validated = validate_synthesis(parsed, case)
|
||||
validation = {"valid": True, "error": None}
|
||||
evaluation = evaluate_synthesis(validated, case["expected"])
|
||||
except requests.RequestException:
|
||||
raise
|
||||
except (json.JSONDecodeError, SynthesisValidationError, ValueError) as exc:
|
||||
validation = {
|
||||
"valid": False,
|
||||
"error_type": type(exc).__name__,
|
||||
"error": str(exc),
|
||||
}
|
||||
evaluation = {
|
||||
"verdict": "FAIL",
|
||||
"reason": f"Schema validation failed: {exc}",
|
||||
"passed_checks": 0,
|
||||
"check_count": 0,
|
||||
"critical_failures": ["schema_validation"],
|
||||
"checks": [],
|
||||
}
|
||||
(case_dir / "validation_result.json").write_text(
|
||||
json.dumps(validation, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
result = {
|
||||
"case_id": case["case_id"],
|
||||
"description": case["description"],
|
||||
**evaluation,
|
||||
"elapsed_seconds": round(time.perf_counter() - started, 3),
|
||||
}
|
||||
(case_dir / "evaluation_result.json").write_text(
|
||||
json.dumps(result, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
|
||||
cases = load_fixture(args.fixture)
|
||||
args.output.mkdir(parents=True, exist_ok=False)
|
||||
started = time.perf_counter()
|
||||
results: list[dict[str, Any]] = []
|
||||
for index, case in enumerate(cases, start=1):
|
||||
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
|
||||
results.append(
|
||||
run_case(
|
||||
case,
|
||||
args.output,
|
||||
args.endpoint,
|
||||
args.model,
|
||||
args.timeout,
|
||||
args.num_ctx,
|
||||
args.num_predict,
|
||||
)
|
||||
)
|
||||
summary = {
|
||||
"experiment": "semantic_synthesis_isolation",
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"model": args.model,
|
||||
"temperature": 0,
|
||||
"think": False,
|
||||
"num_ctx": args.num_ctx,
|
||||
"num_predict": args.num_predict,
|
||||
"case_count": len(cases),
|
||||
"llm_call_count": len(results),
|
||||
"runtime_seconds": round(time.perf_counter() - started, 3),
|
||||
"verdict_counts": {
|
||||
verdict: sum(result["verdict"] == verdict for result in results)
|
||||
for verdict in ("PASS", "PARTIAL", "FAIL")
|
||||
},
|
||||
"results": results,
|
||||
}
|
||||
(args.output / "summary.json").write_text(
|
||||
json.dumps(summary, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
return summary
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
try:
|
||||
summary = run_experiment(args)
|
||||
except (OSError, ValueError, requests.RequestException) as exc:
|
||||
print(f"Error: {exc}")
|
||||
return 1
|
||||
print(json.dumps(summary["verdict_counts"], sort_keys=True))
|
||||
print(f"Artifacts: {args.output.resolve()}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user