Add evidence-near semantic architecture experiments

Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
This commit is contained in:
2026-08-19 15:46:22 +02:00
parent bcb197a908
commit 18beb3385f
29 changed files with 4542 additions and 0 deletions
@@ -0,0 +1,640 @@
#!/usr/bin/env python3
"""Run semantic synthesis with subject detection and evidence assignment fixed."""
from __future__ import annotations
import argparse
import json
import time
from pathlib import Path
from typing import Any
import requests
SCHEMA_VERSION = "experimental-semantic-synthesis-v1"
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
DEFAULT_MODEL = "qwen3.5:9B"
DEFAULT_TIMEOUT = 300
DEFAULT_NUM_CTX = 8192
DEFAULT_NUM_PREDICT = 2048
EVENT_TYPES = {
"idea",
"option",
"proposal",
"objection",
"supporting_argument",
"clarification",
"rejection",
"scoped_acceptance",
"fact",
"technical_finding",
}
OUTCOME_STATUSES = {"established", "rejected", "scoped_acceptance", "tentative"}
class SynthesisValidationError(ValueError):
"""Raised when isolated semantic synthesis output is structurally invalid."""
PROMPT_TEMPLATE = """You perform semantic synthesis for one already known discussion subject.
The subject boundary and evidence assignment are fixed and complete. Do not discover,
split, merge, rename, or omit the subject. Do not assign evidence to another subject.
Interpret only what the supplied evidence semantically establishes.
Semantic distinctions:
- idea: mentioned possibility without stronger commitment
- option: alternative considered without commitment
- proposal: suggested course of action not yet established as work
- objection: argument or concern against something; not automatically unresolved
- rejection: an alternative is explicitly rejected
- scoped_acceptance: accepted only for the stated test, trial, condition, or scope
- proposal is not an action
- no decision is not a tentative decision
- mention is not an unresolved issue
- an action requires explicit assignment, acceptance, commitment, or established work
- an unresolved issue requires a concrete need explicitly left unresolved
Preserve explicit rejection, explicit accepted work, explicit unresolved questions,
and all limits on an outcome. Never generalize trial acceptance into final acceptance.
Use only supplied evidence IDs. Keep concise semantic text in the evidence language.
Return exactly one JSON object. Always include these fields:
{{
"schema_version": "experimental-semantic-synthesis-v1",
"subject_id": "copy the supplied subject_id exactly",
"subject": "copy the supplied subject exactly",
"events": [
{{
"type": "idea|option|proposal|objection|supporting_argument|clarification|rejection|scoped_acceptance|fact|technical_finding",
"text": "supported semantic event",
"evidence_ids": ["e1"]
}}
],
"actions": [
{{
"text": "established action",
"responsible": null,
"due": null,
"evidence_ids": ["e2"]
}}
],
"unresolved_issues": [
{{
"text": "explicitly unresolved issue",
"evidence_ids": ["e3"]
}}
]
}}
The three arrays are structurally required; use [] when none exist.
Add "outcome" only when an outcome was actually established:
{{
"status": "established|rejected|scoped_acceptance|tentative",
"text": "what was actually established",
"scope": "the exact scope, condition, or limit",
"evidence_ids": ["e2"]
}}
Omit outcome completely when there is none. Never use null for outcome. Never use the
string "null"; use JSON null only for unknown responsible or due values.
Fixed Gold input:
{input_json}
"""
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Run the isolated semantic-synthesis Gold experiment."
)
parser.add_argument("fixture", type=Path, help="Fixed-subject Gold bundle JSON.")
parser.add_argument("-o", "--output", type=Path, required=True)
parser.add_argument("--model", default=DEFAULT_MODEL)
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT)
parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX)
parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT)
return parser.parse_args()
def _exact_keys(
value: dict[str, Any], required: set[str], optional: set[str], location: str
) -> None:
missing = required - value.keys()
unknown = value.keys() - required - optional
if missing:
raise SynthesisValidationError(
f"{location} missing required keys: {sorted(missing)}"
)
if unknown:
raise SynthesisValidationError(
f"{location} has unknown keys: {sorted(unknown)}"
)
def _text(value: Any, location: str) -> str:
if not isinstance(value, str) or not value.strip():
raise SynthesisValidationError(f"{location} must be a non-empty string")
return value.strip()
def validate_bundle(case: Any) -> dict[str, Any]:
if not isinstance(case, dict):
raise SynthesisValidationError("case must be an object")
_exact_keys(
case,
{
"case_id",
"description",
"subject_id",
"subject",
"evidence",
"allowed_responsible",
"expected",
},
set(),
"case",
)
_text(case["case_id"], "case.case_id")
_text(case["description"], "case.description")
_text(case["subject_id"], "case.subject_id")
_text(case["subject"], "case.subject")
evidence = case["evidence"]
if not isinstance(evidence, list) or not evidence:
raise SynthesisValidationError("case.evidence must be a non-empty list")
seen: set[str] = set()
for index, item in enumerate(evidence):
location = f"case.evidence[{index}]"
if not isinstance(item, dict):
raise SynthesisValidationError(f"{location} must be an object")
_exact_keys(item, {"evidence_id", "text"}, set(), location)
evidence_id = _text(item["evidence_id"], f"{location}.evidence_id")
if evidence_id in seen:
raise SynthesisValidationError(f"duplicate evidence ID: {evidence_id}")
seen.add(evidence_id)
_text(item["text"], f"{location}.text")
allowed = case["allowed_responsible"]
if not isinstance(allowed, list) or any(
not isinstance(value, str) or not value.strip() for value in allowed
):
raise SynthesisValidationError(
"case.allowed_responsible must be a list of non-empty strings"
)
if len(set(allowed)) != len(allowed):
raise SynthesisValidationError("case.allowed_responsible contains duplicates")
if not isinstance(case["expected"], dict):
raise SynthesisValidationError("case.expected must be an object")
return case
def _evidence_ids(value: Any, location: str, known: set[str]) -> list[str]:
if not isinstance(value, list) or not value:
raise SynthesisValidationError(f"{location} must be a non-empty list")
result: list[str] = []
for index, evidence_id in enumerate(value):
evidence_id = _text(evidence_id, f"{location}[{index}]")
if evidence_id not in known:
raise SynthesisValidationError(
f"{location}[{index}] references unknown evidence ID: {evidence_id}"
)
if evidence_id in result:
raise SynthesisValidationError(
f"{location} contains duplicate evidence ID: {evidence_id}"
)
result.append(evidence_id)
return result
def _nullable_text(value: Any, location: str) -> str | None:
if value is None:
return None
result = _text(value, location)
if result.casefold() == "null":
raise SynthesisValidationError(
f"{location} must use JSON null, not the string 'null'"
)
return result
def validate_synthesis(data: Any, case: dict[str, Any]) -> dict[str, Any]:
validate_bundle(case)
if not isinstance(data, dict):
raise SynthesisValidationError("output must be an object")
_exact_keys(
data,
{
"schema_version",
"subject_id",
"subject",
"events",
"actions",
"unresolved_issues",
},
{"outcome"},
"output",
)
if data["schema_version"] != SCHEMA_VERSION:
raise SynthesisValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
if data["subject_id"] != case["subject_id"]:
raise SynthesisValidationError("model changed fixed subject_id")
if data["subject"] != case["subject"]:
raise SynthesisValidationError("model changed fixed subject")
known = {item["evidence_id"] for item in case["evidence"]}
events = data["events"]
if not isinstance(events, list):
raise SynthesisValidationError("output.events must be an array")
for index, event in enumerate(events):
location = f"output.events[{index}]"
if not isinstance(event, dict):
raise SynthesisValidationError(f"{location} must be an object")
_exact_keys(event, {"type", "text", "evidence_ids"}, set(), location)
if event["type"] not in EVENT_TYPES:
raise SynthesisValidationError(f"{location}.type is invalid")
_text(event["text"], f"{location}.text")
_evidence_ids(event["evidence_ids"], f"{location}.evidence_ids", known)
if "outcome" in data:
outcome = data["outcome"]
if not isinstance(outcome, dict):
raise SynthesisValidationError(
"output.outcome must be an object when present; omit it when absent"
)
_exact_keys(
outcome, {"status", "text", "scope", "evidence_ids"}, set(), "output.outcome"
)
if outcome["status"] not in OUTCOME_STATUSES:
raise SynthesisValidationError("output.outcome.status is invalid")
_text(outcome["text"], "output.outcome.text")
_text(outcome["scope"], "output.outcome.scope")
_evidence_ids(outcome["evidence_ids"], "output.outcome.evidence_ids", known)
actions = data["actions"]
if not isinstance(actions, list):
raise SynthesisValidationError("output.actions must be an array")
allowed = set(case["allowed_responsible"])
for index, action in enumerate(actions):
location = f"output.actions[{index}]"
if not isinstance(action, dict):
raise SynthesisValidationError(f"{location} must be an object")
_exact_keys(
action,
{"text", "responsible", "due", "evidence_ids"},
set(),
location,
)
_text(action["text"], f"{location}.text")
responsible = _nullable_text(action["responsible"], f"{location}.responsible")
if responsible is not None and responsible not in allowed:
raise SynthesisValidationError(
f"{location}.responsible is not allowed: {responsible}"
)
_nullable_text(action["due"], f"{location}.due")
_evidence_ids(action["evidence_ids"], f"{location}.evidence_ids", known)
issues = data["unresolved_issues"]
if not isinstance(issues, list):
raise SynthesisValidationError("output.unresolved_issues must be an array")
for index, issue in enumerate(issues):
location = f"output.unresolved_issues[{index}]"
if not isinstance(issue, dict):
raise SynthesisValidationError(f"{location} must be an object")
_exact_keys(issue, {"text", "evidence_ids"}, set(), location)
_text(issue["text"], f"{location}.text")
_evidence_ids(issue["evidence_ids"], f"{location}.evidence_ids", known)
return data
def build_prompt(case: dict[str, Any]) -> str:
validate_bundle(case)
model_input = {
"subject_id": case["subject_id"],
"subject": case["subject"],
"evidence": case["evidence"],
}
return PROMPT_TEMPLATE.format(
input_json=json.dumps(model_input, ensure_ascii=False, indent=2)
)
def parse_model_json(raw_text: str) -> dict[str, Any]:
data = json.loads(raw_text)
if not isinstance(data, dict):
raise SynthesisValidationError("model response JSON must be an object")
return data
def build_ollama_payload(
model: str, prompt: str, num_ctx: int, num_predict: int
) -> dict[str, Any]:
return {
"model": model,
"prompt": prompt,
"think": False,
"stream": False,
"format": "json",
"options": {
"temperature": 0,
"num_ctx": num_ctx,
"num_predict": num_predict,
},
}
def call_ollama(
endpoint: str,
model: str,
prompt: str,
timeout: int,
num_ctx: int,
num_predict: int,
) -> tuple[str, dict[str, Any]]:
payload = build_ollama_payload(model, prompt, num_ctx, num_predict)
started = time.perf_counter()
response = requests.post(endpoint, json=payload, timeout=timeout)
elapsed = time.perf_counter() - started
response.raise_for_status()
body = response.json()
if not isinstance(body, dict):
raise ValueError("Ollama response must be an object")
raw_text = body.get("response")
if not isinstance(raw_text, str) or not raw_text.strip():
raise ValueError("Ollama returned no usable response text")
metadata = {
"model": body.get("model", model),
"elapsed_seconds": round(elapsed, 3),
"total_duration_ns": body.get("total_duration"),
"load_duration_ns": body.get("load_duration"),
"prompt_eval_count": body.get("prompt_eval_count"),
"prompt_eval_duration_ns": body.get("prompt_eval_duration"),
"eval_count": body.get("eval_count"),
"eval_duration_ns": body.get("eval_duration"),
"configuration": {
"temperature": 0,
"think": False,
"num_ctx": num_ctx,
"num_predict": num_predict,
},
}
return raw_text.strip(), metadata
def _contains(text: str, terms: list[str]) -> bool:
folded = text.casefold()
return any(term.casefold() in folded for term in terms)
def _refs_cover(items: list[dict[str, Any]], expected: list[str]) -> bool:
actual = {
evidence_id
for item in items
for evidence_id in item.get("evidence_ids", [])
}
return set(expected).issubset(actual)
def evaluate_synthesis(data: dict[str, Any], expected: dict[str, Any]) -> dict[str, Any]:
checks: list[dict[str, Any]] = []
def add(name: str, passed: bool, critical: bool = False) -> None:
checks.append({"name": name, "passed": passed, "critical": critical})
events = data["events"]
event_types = [item["type"] for item in events]
for event_type, minimum in expected.get("event_type_minimums", {}).items():
add(f"event:{event_type}", event_types.count(event_type) >= minimum)
allowed_types = set(expected.get("allowed_event_types", EVENT_TYPES))
add("no_unexpected_event_types", set(event_types).issubset(allowed_types))
add(
"event_evidence",
_refs_cover(events, expected.get("event_evidence_ids", [])),
)
outcome_expected = expected["outcome"]
outcome = data.get("outcome")
add(
"outcome_presence",
(outcome is not None) == outcome_expected["required"],
critical=True,
)
if outcome_expected["required"] and outcome is not None:
add("outcome_status", outcome["status"] in outcome_expected["statuses"])
combined = f"{outcome['text']} {outcome['scope']}"
add("outcome_meaning", _contains(combined, outcome_expected["terms"]))
add(
"outcome_scope",
_contains(combined, outcome_expected["scope_terms"]),
critical=True,
)
add(
"outcome_evidence",
set(outcome_expected["evidence_ids"]).issubset(outcome["evidence_ids"]),
critical=True,
)
actions = data["actions"]
expected_actions = expected["actions"]
add(
"action_count",
len(actions) == expected_actions["count"],
critical=True,
)
if expected_actions["count"] and actions:
action_text = " ".join(item["text"] for item in actions)
add("action_meaning", _contains(action_text, expected_actions["terms"]))
if "responsible" in expected_actions:
add(
"action_responsible",
any(item["responsible"] == expected_actions["responsible"] for item in actions),
critical=True,
)
if expected_actions.get("due_terms"):
due_text = " ".join(str(item["due"] or "") for item in actions)
add("action_due", _contains(due_text, expected_actions["due_terms"]))
add(
"action_evidence",
_refs_cover(actions, expected_actions["evidence_ids"]),
critical=True,
)
issues = data["unresolved_issues"]
expected_issues = expected["unresolved_issues"]
add(
"unresolved_count",
len(issues) == expected_issues["count"],
critical=True,
)
if expected_issues["count"] and issues:
issue_text = " ".join(item["text"] for item in issues)
add("unresolved_meaning", _contains(issue_text, expected_issues["terms"]))
add(
"unresolved_evidence",
_refs_cover(issues, expected_issues["evidence_ids"]),
critical=True,
)
passed = sum(item["passed"] for item in checks)
critical_failures = [
item["name"] for item in checks if item["critical"] and not item["passed"]
]
ratio = passed / len(checks)
if ratio == 1:
verdict = "PASS"
elif ratio >= 0.7 and not critical_failures:
verdict = "PARTIAL"
else:
verdict = "FAIL"
failed = [item["name"] for item in checks if not item["passed"]]
return {
"verdict": verdict,
"reason": "All semantic checks passed." if not failed else "Failed: " + ", ".join(failed),
"passed_checks": passed,
"check_count": len(checks),
"critical_failures": critical_failures,
"checks": checks,
}
def load_fixture(path: Path) -> list[dict[str, Any]]:
data = json.loads(path.read_text(encoding="utf-8-sig"))
if not isinstance(data, dict) or set(data) != {"cases"}:
raise SynthesisValidationError("fixture must contain exactly a cases list")
cases = data["cases"]
if not isinstance(cases, list) or not cases:
raise SynthesisValidationError("fixture cases must be a non-empty list")
seen: set[str] = set()
for case in cases:
validate_bundle(case)
if case["case_id"] in seen:
raise SynthesisValidationError(f"duplicate case ID: {case['case_id']}")
seen.add(case["case_id"])
return cases
def run_case(
case: dict[str, Any],
output_root: Path,
endpoint: str,
model: str,
timeout: int,
num_ctx: int,
num_predict: int,
) -> dict[str, Any]:
case_dir = output_root / case["case_id"]
case_dir.mkdir(parents=True, exist_ok=False)
gold_input = {
"case_id": case["case_id"],
"description": case["description"],
"subject_id": case["subject_id"],
"subject": case["subject"],
"evidence": case["evidence"],
}
(case_dir / "gold_input.json").write_text(
json.dumps(gold_input, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
prompt = build_prompt(case)
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
started = time.perf_counter()
try:
raw_text, metadata = call_ollama(
endpoint, model, prompt, timeout, num_ctx, num_predict
)
(case_dir / "raw_model_response.txt").write_text(raw_text + "\n", encoding="utf-8")
(case_dir / "ollama_metadata.json").write_text(
json.dumps(metadata, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
parsed = parse_model_json(raw_text)
(case_dir / "parsed_response.json").write_text(
json.dumps(parsed, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
validated = validate_synthesis(parsed, case)
validation = {"valid": True, "error": None}
evaluation = evaluate_synthesis(validated, case["expected"])
except requests.RequestException:
raise
except (json.JSONDecodeError, SynthesisValidationError, ValueError) as exc:
validation = {
"valid": False,
"error_type": type(exc).__name__,
"error": str(exc),
}
evaluation = {
"verdict": "FAIL",
"reason": f"Schema validation failed: {exc}",
"passed_checks": 0,
"check_count": 0,
"critical_failures": ["schema_validation"],
"checks": [],
}
(case_dir / "validation_result.json").write_text(
json.dumps(validation, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
result = {
"case_id": case["case_id"],
"description": case["description"],
**evaluation,
"elapsed_seconds": round(time.perf_counter() - started, 3),
}
(case_dir / "evaluation_result.json").write_text(
json.dumps(result, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
return result
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
cases = load_fixture(args.fixture)
args.output.mkdir(parents=True, exist_ok=False)
started = time.perf_counter()
results: list[dict[str, Any]] = []
for index, case in enumerate(cases, start=1):
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
results.append(
run_case(
case,
args.output,
args.endpoint,
args.model,
args.timeout,
args.num_ctx,
args.num_predict,
)
)
summary = {
"experiment": "semantic_synthesis_isolation",
"schema_version": SCHEMA_VERSION,
"model": args.model,
"temperature": 0,
"think": False,
"num_ctx": args.num_ctx,
"num_predict": args.num_predict,
"case_count": len(cases),
"llm_call_count": len(results),
"runtime_seconds": round(time.perf_counter() - started, 3),
"verdict_counts": {
verdict: sum(result["verdict"] == verdict for result in results)
for verdict in ("PASS", "PARTIAL", "FAIL")
},
"results": results,
}
(args.output / "summary.json").write_text(
json.dumps(summary, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
return summary
def main() -> int:
args = parse_args()
try:
summary = run_experiment(args)
except (OSError, ValueError, requests.RequestException) as exc:
print(f"Error: {exc}")
return 1
print(json.dumps(summary["verdict_counts"], sort_keys=True))
print(f"Artifacts: {args.output.resolve()}")
return 0
if __name__ == "__main__":
raise SystemExit(main())