Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
641 lines
23 KiB
Python
641 lines
23 KiB
Python
#!/usr/bin/env python3
|
|
"""Run semantic synthesis with subject detection and evidence assignment fixed."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import time
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
import requests
|
|
|
|
|
|
SCHEMA_VERSION = "experimental-semantic-synthesis-v1"
|
|
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
|
|
DEFAULT_MODEL = "qwen3.5:9B"
|
|
DEFAULT_TIMEOUT = 300
|
|
DEFAULT_NUM_CTX = 8192
|
|
DEFAULT_NUM_PREDICT = 2048
|
|
|
|
EVENT_TYPES = {
|
|
"idea",
|
|
"option",
|
|
"proposal",
|
|
"objection",
|
|
"supporting_argument",
|
|
"clarification",
|
|
"rejection",
|
|
"scoped_acceptance",
|
|
"fact",
|
|
"technical_finding",
|
|
}
|
|
OUTCOME_STATUSES = {"established", "rejected", "scoped_acceptance", "tentative"}
|
|
|
|
|
|
class SynthesisValidationError(ValueError):
|
|
"""Raised when isolated semantic synthesis output is structurally invalid."""
|
|
|
|
|
|
PROMPT_TEMPLATE = """You perform semantic synthesis for one already known discussion subject.
|
|
|
|
The subject boundary and evidence assignment are fixed and complete. Do not discover,
|
|
split, merge, rename, or omit the subject. Do not assign evidence to another subject.
|
|
Interpret only what the supplied evidence semantically establishes.
|
|
|
|
Semantic distinctions:
|
|
- idea: mentioned possibility without stronger commitment
|
|
- option: alternative considered without commitment
|
|
- proposal: suggested course of action not yet established as work
|
|
- objection: argument or concern against something; not automatically unresolved
|
|
- rejection: an alternative is explicitly rejected
|
|
- scoped_acceptance: accepted only for the stated test, trial, condition, or scope
|
|
- proposal is not an action
|
|
- no decision is not a tentative decision
|
|
- mention is not an unresolved issue
|
|
- an action requires explicit assignment, acceptance, commitment, or established work
|
|
- an unresolved issue requires a concrete need explicitly left unresolved
|
|
|
|
Preserve explicit rejection, explicit accepted work, explicit unresolved questions,
|
|
and all limits on an outcome. Never generalize trial acceptance into final acceptance.
|
|
Use only supplied evidence IDs. Keep concise semantic text in the evidence language.
|
|
|
|
Return exactly one JSON object. Always include these fields:
|
|
{{
|
|
"schema_version": "experimental-semantic-synthesis-v1",
|
|
"subject_id": "copy the supplied subject_id exactly",
|
|
"subject": "copy the supplied subject exactly",
|
|
"events": [
|
|
{{
|
|
"type": "idea|option|proposal|objection|supporting_argument|clarification|rejection|scoped_acceptance|fact|technical_finding",
|
|
"text": "supported semantic event",
|
|
"evidence_ids": ["e1"]
|
|
}}
|
|
],
|
|
"actions": [
|
|
{{
|
|
"text": "established action",
|
|
"responsible": null,
|
|
"due": null,
|
|
"evidence_ids": ["e2"]
|
|
}}
|
|
],
|
|
"unresolved_issues": [
|
|
{{
|
|
"text": "explicitly unresolved issue",
|
|
"evidence_ids": ["e3"]
|
|
}}
|
|
]
|
|
}}
|
|
|
|
The three arrays are structurally required; use [] when none exist.
|
|
Add "outcome" only when an outcome was actually established:
|
|
{{
|
|
"status": "established|rejected|scoped_acceptance|tentative",
|
|
"text": "what was actually established",
|
|
"scope": "the exact scope, condition, or limit",
|
|
"evidence_ids": ["e2"]
|
|
}}
|
|
Omit outcome completely when there is none. Never use null for outcome. Never use the
|
|
string "null"; use JSON null only for unknown responsible or due values.
|
|
|
|
Fixed Gold input:
|
|
{input_json}
|
|
"""
|
|
|
|
|
|
def parse_args() -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(
|
|
description="Run the isolated semantic-synthesis Gold experiment."
|
|
)
|
|
parser.add_argument("fixture", type=Path, help="Fixed-subject Gold bundle JSON.")
|
|
parser.add_argument("-o", "--output", type=Path, required=True)
|
|
parser.add_argument("--model", default=DEFAULT_MODEL)
|
|
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
|
|
parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT)
|
|
parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX)
|
|
parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT)
|
|
return parser.parse_args()
|
|
|
|
|
|
def _exact_keys(
|
|
value: dict[str, Any], required: set[str], optional: set[str], location: str
|
|
) -> None:
|
|
missing = required - value.keys()
|
|
unknown = value.keys() - required - optional
|
|
if missing:
|
|
raise SynthesisValidationError(
|
|
f"{location} missing required keys: {sorted(missing)}"
|
|
)
|
|
if unknown:
|
|
raise SynthesisValidationError(
|
|
f"{location} has unknown keys: {sorted(unknown)}"
|
|
)
|
|
|
|
|
|
def _text(value: Any, location: str) -> str:
|
|
if not isinstance(value, str) or not value.strip():
|
|
raise SynthesisValidationError(f"{location} must be a non-empty string")
|
|
return value.strip()
|
|
|
|
|
|
def validate_bundle(case: Any) -> dict[str, Any]:
|
|
if not isinstance(case, dict):
|
|
raise SynthesisValidationError("case must be an object")
|
|
_exact_keys(
|
|
case,
|
|
{
|
|
"case_id",
|
|
"description",
|
|
"subject_id",
|
|
"subject",
|
|
"evidence",
|
|
"allowed_responsible",
|
|
"expected",
|
|
},
|
|
set(),
|
|
"case",
|
|
)
|
|
_text(case["case_id"], "case.case_id")
|
|
_text(case["description"], "case.description")
|
|
_text(case["subject_id"], "case.subject_id")
|
|
_text(case["subject"], "case.subject")
|
|
evidence = case["evidence"]
|
|
if not isinstance(evidence, list) or not evidence:
|
|
raise SynthesisValidationError("case.evidence must be a non-empty list")
|
|
seen: set[str] = set()
|
|
for index, item in enumerate(evidence):
|
|
location = f"case.evidence[{index}]"
|
|
if not isinstance(item, dict):
|
|
raise SynthesisValidationError(f"{location} must be an object")
|
|
_exact_keys(item, {"evidence_id", "text"}, set(), location)
|
|
evidence_id = _text(item["evidence_id"], f"{location}.evidence_id")
|
|
if evidence_id in seen:
|
|
raise SynthesisValidationError(f"duplicate evidence ID: {evidence_id}")
|
|
seen.add(evidence_id)
|
|
_text(item["text"], f"{location}.text")
|
|
allowed = case["allowed_responsible"]
|
|
if not isinstance(allowed, list) or any(
|
|
not isinstance(value, str) or not value.strip() for value in allowed
|
|
):
|
|
raise SynthesisValidationError(
|
|
"case.allowed_responsible must be a list of non-empty strings"
|
|
)
|
|
if len(set(allowed)) != len(allowed):
|
|
raise SynthesisValidationError("case.allowed_responsible contains duplicates")
|
|
if not isinstance(case["expected"], dict):
|
|
raise SynthesisValidationError("case.expected must be an object")
|
|
return case
|
|
|
|
|
|
def _evidence_ids(value: Any, location: str, known: set[str]) -> list[str]:
|
|
if not isinstance(value, list) or not value:
|
|
raise SynthesisValidationError(f"{location} must be a non-empty list")
|
|
result: list[str] = []
|
|
for index, evidence_id in enumerate(value):
|
|
evidence_id = _text(evidence_id, f"{location}[{index}]")
|
|
if evidence_id not in known:
|
|
raise SynthesisValidationError(
|
|
f"{location}[{index}] references unknown evidence ID: {evidence_id}"
|
|
)
|
|
if evidence_id in result:
|
|
raise SynthesisValidationError(
|
|
f"{location} contains duplicate evidence ID: {evidence_id}"
|
|
)
|
|
result.append(evidence_id)
|
|
return result
|
|
|
|
|
|
def _nullable_text(value: Any, location: str) -> str | None:
|
|
if value is None:
|
|
return None
|
|
result = _text(value, location)
|
|
if result.casefold() == "null":
|
|
raise SynthesisValidationError(
|
|
f"{location} must use JSON null, not the string 'null'"
|
|
)
|
|
return result
|
|
|
|
|
|
def validate_synthesis(data: Any, case: dict[str, Any]) -> dict[str, Any]:
|
|
validate_bundle(case)
|
|
if not isinstance(data, dict):
|
|
raise SynthesisValidationError("output must be an object")
|
|
_exact_keys(
|
|
data,
|
|
{
|
|
"schema_version",
|
|
"subject_id",
|
|
"subject",
|
|
"events",
|
|
"actions",
|
|
"unresolved_issues",
|
|
},
|
|
{"outcome"},
|
|
"output",
|
|
)
|
|
if data["schema_version"] != SCHEMA_VERSION:
|
|
raise SynthesisValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
|
|
if data["subject_id"] != case["subject_id"]:
|
|
raise SynthesisValidationError("model changed fixed subject_id")
|
|
if data["subject"] != case["subject"]:
|
|
raise SynthesisValidationError("model changed fixed subject")
|
|
|
|
known = {item["evidence_id"] for item in case["evidence"]}
|
|
events = data["events"]
|
|
if not isinstance(events, list):
|
|
raise SynthesisValidationError("output.events must be an array")
|
|
for index, event in enumerate(events):
|
|
location = f"output.events[{index}]"
|
|
if not isinstance(event, dict):
|
|
raise SynthesisValidationError(f"{location} must be an object")
|
|
_exact_keys(event, {"type", "text", "evidence_ids"}, set(), location)
|
|
if event["type"] not in EVENT_TYPES:
|
|
raise SynthesisValidationError(f"{location}.type is invalid")
|
|
_text(event["text"], f"{location}.text")
|
|
_evidence_ids(event["evidence_ids"], f"{location}.evidence_ids", known)
|
|
|
|
if "outcome" in data:
|
|
outcome = data["outcome"]
|
|
if not isinstance(outcome, dict):
|
|
raise SynthesisValidationError(
|
|
"output.outcome must be an object when present; omit it when absent"
|
|
)
|
|
_exact_keys(
|
|
outcome, {"status", "text", "scope", "evidence_ids"}, set(), "output.outcome"
|
|
)
|
|
if outcome["status"] not in OUTCOME_STATUSES:
|
|
raise SynthesisValidationError("output.outcome.status is invalid")
|
|
_text(outcome["text"], "output.outcome.text")
|
|
_text(outcome["scope"], "output.outcome.scope")
|
|
_evidence_ids(outcome["evidence_ids"], "output.outcome.evidence_ids", known)
|
|
|
|
actions = data["actions"]
|
|
if not isinstance(actions, list):
|
|
raise SynthesisValidationError("output.actions must be an array")
|
|
allowed = set(case["allowed_responsible"])
|
|
for index, action in enumerate(actions):
|
|
location = f"output.actions[{index}]"
|
|
if not isinstance(action, dict):
|
|
raise SynthesisValidationError(f"{location} must be an object")
|
|
_exact_keys(
|
|
action,
|
|
{"text", "responsible", "due", "evidence_ids"},
|
|
set(),
|
|
location,
|
|
)
|
|
_text(action["text"], f"{location}.text")
|
|
responsible = _nullable_text(action["responsible"], f"{location}.responsible")
|
|
if responsible is not None and responsible not in allowed:
|
|
raise SynthesisValidationError(
|
|
f"{location}.responsible is not allowed: {responsible}"
|
|
)
|
|
_nullable_text(action["due"], f"{location}.due")
|
|
_evidence_ids(action["evidence_ids"], f"{location}.evidence_ids", known)
|
|
|
|
issues = data["unresolved_issues"]
|
|
if not isinstance(issues, list):
|
|
raise SynthesisValidationError("output.unresolved_issues must be an array")
|
|
for index, issue in enumerate(issues):
|
|
location = f"output.unresolved_issues[{index}]"
|
|
if not isinstance(issue, dict):
|
|
raise SynthesisValidationError(f"{location} must be an object")
|
|
_exact_keys(issue, {"text", "evidence_ids"}, set(), location)
|
|
_text(issue["text"], f"{location}.text")
|
|
_evidence_ids(issue["evidence_ids"], f"{location}.evidence_ids", known)
|
|
return data
|
|
|
|
|
|
def build_prompt(case: dict[str, Any]) -> str:
|
|
validate_bundle(case)
|
|
model_input = {
|
|
"subject_id": case["subject_id"],
|
|
"subject": case["subject"],
|
|
"evidence": case["evidence"],
|
|
}
|
|
return PROMPT_TEMPLATE.format(
|
|
input_json=json.dumps(model_input, ensure_ascii=False, indent=2)
|
|
)
|
|
|
|
|
|
def parse_model_json(raw_text: str) -> dict[str, Any]:
|
|
data = json.loads(raw_text)
|
|
if not isinstance(data, dict):
|
|
raise SynthesisValidationError("model response JSON must be an object")
|
|
return data
|
|
|
|
|
|
def build_ollama_payload(
|
|
model: str, prompt: str, num_ctx: int, num_predict: int
|
|
) -> dict[str, Any]:
|
|
return {
|
|
"model": model,
|
|
"prompt": prompt,
|
|
"think": False,
|
|
"stream": False,
|
|
"format": "json",
|
|
"options": {
|
|
"temperature": 0,
|
|
"num_ctx": num_ctx,
|
|
"num_predict": num_predict,
|
|
},
|
|
}
|
|
|
|
|
|
def call_ollama(
|
|
endpoint: str,
|
|
model: str,
|
|
prompt: str,
|
|
timeout: int,
|
|
num_ctx: int,
|
|
num_predict: int,
|
|
) -> tuple[str, dict[str, Any]]:
|
|
payload = build_ollama_payload(model, prompt, num_ctx, num_predict)
|
|
started = time.perf_counter()
|
|
response = requests.post(endpoint, json=payload, timeout=timeout)
|
|
elapsed = time.perf_counter() - started
|
|
response.raise_for_status()
|
|
body = response.json()
|
|
if not isinstance(body, dict):
|
|
raise ValueError("Ollama response must be an object")
|
|
raw_text = body.get("response")
|
|
if not isinstance(raw_text, str) or not raw_text.strip():
|
|
raise ValueError("Ollama returned no usable response text")
|
|
metadata = {
|
|
"model": body.get("model", model),
|
|
"elapsed_seconds": round(elapsed, 3),
|
|
"total_duration_ns": body.get("total_duration"),
|
|
"load_duration_ns": body.get("load_duration"),
|
|
"prompt_eval_count": body.get("prompt_eval_count"),
|
|
"prompt_eval_duration_ns": body.get("prompt_eval_duration"),
|
|
"eval_count": body.get("eval_count"),
|
|
"eval_duration_ns": body.get("eval_duration"),
|
|
"configuration": {
|
|
"temperature": 0,
|
|
"think": False,
|
|
"num_ctx": num_ctx,
|
|
"num_predict": num_predict,
|
|
},
|
|
}
|
|
return raw_text.strip(), metadata
|
|
|
|
|
|
def _contains(text: str, terms: list[str]) -> bool:
|
|
folded = text.casefold()
|
|
return any(term.casefold() in folded for term in terms)
|
|
|
|
|
|
def _refs_cover(items: list[dict[str, Any]], expected: list[str]) -> bool:
|
|
actual = {
|
|
evidence_id
|
|
for item in items
|
|
for evidence_id in item.get("evidence_ids", [])
|
|
}
|
|
return set(expected).issubset(actual)
|
|
|
|
|
|
def evaluate_synthesis(data: dict[str, Any], expected: dict[str, Any]) -> dict[str, Any]:
|
|
checks: list[dict[str, Any]] = []
|
|
|
|
def add(name: str, passed: bool, critical: bool = False) -> None:
|
|
checks.append({"name": name, "passed": passed, "critical": critical})
|
|
|
|
events = data["events"]
|
|
event_types = [item["type"] for item in events]
|
|
for event_type, minimum in expected.get("event_type_minimums", {}).items():
|
|
add(f"event:{event_type}", event_types.count(event_type) >= minimum)
|
|
allowed_types = set(expected.get("allowed_event_types", EVENT_TYPES))
|
|
add("no_unexpected_event_types", set(event_types).issubset(allowed_types))
|
|
add(
|
|
"event_evidence",
|
|
_refs_cover(events, expected.get("event_evidence_ids", [])),
|
|
)
|
|
|
|
outcome_expected = expected["outcome"]
|
|
outcome = data.get("outcome")
|
|
add(
|
|
"outcome_presence",
|
|
(outcome is not None) == outcome_expected["required"],
|
|
critical=True,
|
|
)
|
|
if outcome_expected["required"] and outcome is not None:
|
|
add("outcome_status", outcome["status"] in outcome_expected["statuses"])
|
|
combined = f"{outcome['text']} {outcome['scope']}"
|
|
add("outcome_meaning", _contains(combined, outcome_expected["terms"]))
|
|
add(
|
|
"outcome_scope",
|
|
_contains(combined, outcome_expected["scope_terms"]),
|
|
critical=True,
|
|
)
|
|
add(
|
|
"outcome_evidence",
|
|
set(outcome_expected["evidence_ids"]).issubset(outcome["evidence_ids"]),
|
|
critical=True,
|
|
)
|
|
|
|
actions = data["actions"]
|
|
expected_actions = expected["actions"]
|
|
add(
|
|
"action_count",
|
|
len(actions) == expected_actions["count"],
|
|
critical=True,
|
|
)
|
|
if expected_actions["count"] and actions:
|
|
action_text = " ".join(item["text"] for item in actions)
|
|
add("action_meaning", _contains(action_text, expected_actions["terms"]))
|
|
if "responsible" in expected_actions:
|
|
add(
|
|
"action_responsible",
|
|
any(item["responsible"] == expected_actions["responsible"] for item in actions),
|
|
critical=True,
|
|
)
|
|
if expected_actions.get("due_terms"):
|
|
due_text = " ".join(str(item["due"] or "") for item in actions)
|
|
add("action_due", _contains(due_text, expected_actions["due_terms"]))
|
|
add(
|
|
"action_evidence",
|
|
_refs_cover(actions, expected_actions["evidence_ids"]),
|
|
critical=True,
|
|
)
|
|
|
|
issues = data["unresolved_issues"]
|
|
expected_issues = expected["unresolved_issues"]
|
|
add(
|
|
"unresolved_count",
|
|
len(issues) == expected_issues["count"],
|
|
critical=True,
|
|
)
|
|
if expected_issues["count"] and issues:
|
|
issue_text = " ".join(item["text"] for item in issues)
|
|
add("unresolved_meaning", _contains(issue_text, expected_issues["terms"]))
|
|
add(
|
|
"unresolved_evidence",
|
|
_refs_cover(issues, expected_issues["evidence_ids"]),
|
|
critical=True,
|
|
)
|
|
|
|
passed = sum(item["passed"] for item in checks)
|
|
critical_failures = [
|
|
item["name"] for item in checks if item["critical"] and not item["passed"]
|
|
]
|
|
ratio = passed / len(checks)
|
|
if ratio == 1:
|
|
verdict = "PASS"
|
|
elif ratio >= 0.7 and not critical_failures:
|
|
verdict = "PARTIAL"
|
|
else:
|
|
verdict = "FAIL"
|
|
failed = [item["name"] for item in checks if not item["passed"]]
|
|
return {
|
|
"verdict": verdict,
|
|
"reason": "All semantic checks passed." if not failed else "Failed: " + ", ".join(failed),
|
|
"passed_checks": passed,
|
|
"check_count": len(checks),
|
|
"critical_failures": critical_failures,
|
|
"checks": checks,
|
|
}
|
|
|
|
|
|
def load_fixture(path: Path) -> list[dict[str, Any]]:
|
|
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
|
if not isinstance(data, dict) or set(data) != {"cases"}:
|
|
raise SynthesisValidationError("fixture must contain exactly a cases list")
|
|
cases = data["cases"]
|
|
if not isinstance(cases, list) or not cases:
|
|
raise SynthesisValidationError("fixture cases must be a non-empty list")
|
|
seen: set[str] = set()
|
|
for case in cases:
|
|
validate_bundle(case)
|
|
if case["case_id"] in seen:
|
|
raise SynthesisValidationError(f"duplicate case ID: {case['case_id']}")
|
|
seen.add(case["case_id"])
|
|
return cases
|
|
|
|
|
|
def run_case(
|
|
case: dict[str, Any],
|
|
output_root: Path,
|
|
endpoint: str,
|
|
model: str,
|
|
timeout: int,
|
|
num_ctx: int,
|
|
num_predict: int,
|
|
) -> dict[str, Any]:
|
|
case_dir = output_root / case["case_id"]
|
|
case_dir.mkdir(parents=True, exist_ok=False)
|
|
gold_input = {
|
|
"case_id": case["case_id"],
|
|
"description": case["description"],
|
|
"subject_id": case["subject_id"],
|
|
"subject": case["subject"],
|
|
"evidence": case["evidence"],
|
|
}
|
|
(case_dir / "gold_input.json").write_text(
|
|
json.dumps(gold_input, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
)
|
|
prompt = build_prompt(case)
|
|
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
|
|
started = time.perf_counter()
|
|
try:
|
|
raw_text, metadata = call_ollama(
|
|
endpoint, model, prompt, timeout, num_ctx, num_predict
|
|
)
|
|
(case_dir / "raw_model_response.txt").write_text(raw_text + "\n", encoding="utf-8")
|
|
(case_dir / "ollama_metadata.json").write_text(
|
|
json.dumps(metadata, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
)
|
|
parsed = parse_model_json(raw_text)
|
|
(case_dir / "parsed_response.json").write_text(
|
|
json.dumps(parsed, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
)
|
|
validated = validate_synthesis(parsed, case)
|
|
validation = {"valid": True, "error": None}
|
|
evaluation = evaluate_synthesis(validated, case["expected"])
|
|
except requests.RequestException:
|
|
raise
|
|
except (json.JSONDecodeError, SynthesisValidationError, ValueError) as exc:
|
|
validation = {
|
|
"valid": False,
|
|
"error_type": type(exc).__name__,
|
|
"error": str(exc),
|
|
}
|
|
evaluation = {
|
|
"verdict": "FAIL",
|
|
"reason": f"Schema validation failed: {exc}",
|
|
"passed_checks": 0,
|
|
"check_count": 0,
|
|
"critical_failures": ["schema_validation"],
|
|
"checks": [],
|
|
}
|
|
(case_dir / "validation_result.json").write_text(
|
|
json.dumps(validation, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
)
|
|
result = {
|
|
"case_id": case["case_id"],
|
|
"description": case["description"],
|
|
**evaluation,
|
|
"elapsed_seconds": round(time.perf_counter() - started, 3),
|
|
}
|
|
(case_dir / "evaluation_result.json").write_text(
|
|
json.dumps(result, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
)
|
|
return result
|
|
|
|
|
|
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
|
|
cases = load_fixture(args.fixture)
|
|
args.output.mkdir(parents=True, exist_ok=False)
|
|
started = time.perf_counter()
|
|
results: list[dict[str, Any]] = []
|
|
for index, case in enumerate(cases, start=1):
|
|
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
|
|
results.append(
|
|
run_case(
|
|
case,
|
|
args.output,
|
|
args.endpoint,
|
|
args.model,
|
|
args.timeout,
|
|
args.num_ctx,
|
|
args.num_predict,
|
|
)
|
|
)
|
|
summary = {
|
|
"experiment": "semantic_synthesis_isolation",
|
|
"schema_version": SCHEMA_VERSION,
|
|
"model": args.model,
|
|
"temperature": 0,
|
|
"think": False,
|
|
"num_ctx": args.num_ctx,
|
|
"num_predict": args.num_predict,
|
|
"case_count": len(cases),
|
|
"llm_call_count": len(results),
|
|
"runtime_seconds": round(time.perf_counter() - started, 3),
|
|
"verdict_counts": {
|
|
verdict: sum(result["verdict"] == verdict for result in results)
|
|
for verdict in ("PASS", "PARTIAL", "FAIL")
|
|
},
|
|
"results": results,
|
|
}
|
|
(args.output / "summary.json").write_text(
|
|
json.dumps(summary, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
)
|
|
return summary
|
|
|
|
|
|
def main() -> int:
|
|
args = parse_args()
|
|
try:
|
|
summary = run_experiment(args)
|
|
except (OSError, ValueError, requests.RequestException) as exc:
|
|
print(f"Error: {exc}")
|
|
return 1
|
|
print(json.dumps(summary["verdict_counts"], sort_keys=True))
|
|
print(f"Artifacts: {args.output.resolve()}")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|