Add evidence-near semantic architecture experiments

Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
This commit is contained in:
2026-08-19 15:46:22 +02:00
parent bcb197a908
commit 18beb3385f
29 changed files with 4542 additions and 0 deletions
@@ -0,0 +1 @@
"""Isolated evidence-near observation experiment."""
@@ -0,0 +1,395 @@
#!/usr/bin/env python3
"""Extract evidence-near observations for a fixed Discussion Subject."""
from __future__ import annotations
import argparse
import json
import re
import time
from pathlib import Path
from typing import Any
import requests
SCHEMA_VERSION = "experimental-evidence-observations-v1"
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
DEFAULT_MODEL = "qwen3.5:9B"
DEFAULT_TIMEOUT = 300
DEFAULT_NUM_CTX = 16384
DEFAULT_NUM_PREDICT = 4096
RELATIONS = {"none", "supports", "opposes", "qualifies", "limits_scope"}
MODALITIES = {
"factual",
"possible",
"suggested",
"interpersonal_request",
"impersonal_necessity",
"information_question",
"committed",
}
TEMPORALITIES = {"existing", "future", "completed", "unspecified"}
EVALUATIONS = {"positive", "negative", "none"}
AGREEMENTS = {"accepted", "rejected", "unclear", "none"}
RESPONSIBILITIES = {"none", "named", "accepted"}
UNCERTAINTIES = {"present", "absent"}
CLARIFICATION_NEEDS = {"explicit", "implicit", "none"}
OBSERVATION_ID_RE = re.compile(r"^obs_[1-9][0-9]*$")
class ObservationValidationError(ValueError):
"""Raised when an experimental fixture or model output is invalid."""
PROMPT_TEMPLATE = """You extract atomic, evidence-near observations for one fixed Discussion Subject.
Stop before protocol interpretation. Never classify anything as an idea, proposal,
objection, decision, action item, or open question. Do not determine protocol
eligibility, reconstruct topics, generate a protocol, or invent missing stages.
Split an evidence unit into multiple observations when it directly contains multiple
propositions. Preserve every observation's source evidence ID. Use concise content in
the evidence language.
Return exactly one JSON object with this shape:
{{
"schema_version": "experimental-evidence-observations-v1",
"subject_id": "copy exactly",
"subject": "copy exactly",
"observations": [
{{
"observation_id": "obs_1",
"evidence_id": "e1",
"content": "directly supported atomic observation",
"target": "discussion_subject",
"relation": "none",
"modality": "factual",
"temporality": "existing",
"evaluation": "none",
"agreement": "none",
"responsibility": "none",
"person": null,
"uncertainty": "absent",
"clarification_need": "none",
"scope": "absent"
}}
]
}}
Rules:
- Number observation_id sequentially as obs_1, obs_2, ... in evidence order.
- target is "discussion_subject", one earlier observation_id, or a non-empty list of
earlier observation_ids only when the evidence jointly refers to them.
- relation is only none, supports, opposes, qualifies, or limits_scope.
- modality is only factual, possible, suggested, interpersonal_request,
impersonal_necessity, information_question, or committed.
- interpersonal_request is a direct request to another person.
- impersonal_necessity says something needs to happen without assigning it.
- information_question expresses missing information without assigning work.
- temporality is only existing, future, completed, or unspecified.
- evaluation is positive, negative, or none. Do not infer evaluation from world
knowledge. A bare cost or technical fact normally has evaluation none.
- agreement is only accepted, rejected, unclear, or none and applies to target.
- responsibility is none, named, or accepted. Use named only for an explicitly
addressed candidate and accepted only for explicit acceptance/commitment.
- person is the explicit person's name for named/accepted responsibility; otherwise
use JSON null. Mentioning or speaking in first person does not establish ownership.
- uncertainty is present or absent.
- clarification_need is explicit, implicit, or none.
- scope is an evidence-grounded qualifier, or exactly "absent". Never use null or the
string "null" anywhere.
- Confirmation of a rejection targets the rejection observation, not the option.
- A trial-only qualification targets and limits the accepted trial.
- A negative consequence can oppose another observation without requiring
clarification.
- Personal preference is not group rejection.
- Collective "we" does not name an individual owner.
Fixed Gold input:
{input_json}
"""
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(description="Run the evidence-observation Gold experiment.")
parser.add_argument("fixture", type=Path)
parser.add_argument("-o", "--output", type=Path, required=True)
parser.add_argument("--model", default=DEFAULT_MODEL)
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT)
parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX)
parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT)
return parser.parse_args()
def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None:
missing = required - value.keys()
unknown = value.keys() - required
if missing:
raise ObservationValidationError(f"{location} missing required keys: {sorted(missing)}")
if unknown:
raise ObservationValidationError(f"{location} has unknown keys: {sorted(unknown)}")
def _text(value: Any, location: str) -> str:
if not isinstance(value, str) or not value.strip():
raise ObservationValidationError(f"{location} must be a non-empty string")
result = value.strip()
if result.casefold() == "null":
raise ObservationValidationError(f"{location} must not be the string 'null'")
return result
OBSERVATION_KEYS = {
"observation_id", "evidence_id", "content", "target", "relation", "modality",
"temporality", "evaluation", "agreement", "responsibility", "person",
"uncertainty", "clarification_need", "scope",
}
def _validate_target(value: Any, location: str, earlier: set[str]) -> None:
if isinstance(value, str):
target = _text(value, location)
if target != "discussion_subject" and target not in earlier:
raise ObservationValidationError(f"{location} references unknown or later observation: {target}")
return
if not isinstance(value, list) or not value:
raise ObservationValidationError(f"{location} must be discussion_subject, an earlier observation ID, or a non-empty list")
if len(value) < 2:
raise ObservationValidationError(f"{location} list must contain at least two jointly referenced observations")
seen: set[str] = set()
for index, item in enumerate(value):
target = _text(item, f"{location}[{index}]")
if target not in earlier:
raise ObservationValidationError(f"{location}[{index}] references unknown or later observation: {target}")
if target in seen:
raise ObservationValidationError(f"{location} contains duplicate target: {target}")
seen.add(target)
def validate_observations(data: Any, case: dict[str, Any]) -> dict[str, Any]:
validate_case(case)
if not isinstance(data, dict):
raise ObservationValidationError("output must be an object")
_exact_keys(data, {"schema_version", "subject_id", "subject", "observations"}, "output")
if data["schema_version"] != SCHEMA_VERSION:
raise ObservationValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
if data["subject_id"] != case["subject_id"] or data["subject"] != case["subject"]:
raise ObservationValidationError("model changed the fixed Discussion Subject")
observations = data["observations"]
if not isinstance(observations, list) or not observations:
raise ObservationValidationError("output.observations must be a non-empty array")
known_evidence = {item["evidence_id"] for item in case["evidence"]}
earlier: set[str] = set()
for index, observation in enumerate(observations, start=1):
location = f"output.observations[{index - 1}]"
if not isinstance(observation, dict):
raise ObservationValidationError(f"{location} must be an object")
_exact_keys(observation, OBSERVATION_KEYS, location)
observation_id = _text(observation["observation_id"], f"{location}.observation_id")
if not OBSERVATION_ID_RE.fullmatch(observation_id) or observation_id != f"obs_{index}":
raise ObservationValidationError(f"{location}.observation_id must be obs_{index}")
evidence_id = _text(observation["evidence_id"], f"{location}.evidence_id")
if evidence_id not in known_evidence:
raise ObservationValidationError(f"{location}.evidence_id references unknown evidence: {evidence_id}")
_text(observation["content"], f"{location}.content")
_validate_target(observation["target"], f"{location}.target", earlier)
for field, values in (
("relation", RELATIONS), ("modality", MODALITIES),
("temporality", TEMPORALITIES), ("evaluation", EVALUATIONS),
("agreement", AGREEMENTS), ("responsibility", RESPONSIBILITIES),
("uncertainty", UNCERTAINTIES), ("clarification_need", CLARIFICATION_NEEDS),
):
if observation[field] not in values:
raise ObservationValidationError(f"{location}.{field} is invalid: {observation[field]!r}")
person = observation["person"]
if observation["responsibility"] == "none":
if person is not None:
raise ObservationValidationError(f"{location}.person must be JSON null when responsibility is none")
else:
_text(person, f"{location}.person")
scope = _text(observation["scope"], f"{location}.scope")
if scope.casefold() == "null":
raise ObservationValidationError(f"{location}.scope must use 'absent', not 'null'")
earlier.add(observation_id)
return data
def validate_case(case: Any) -> dict[str, Any]:
if not isinstance(case, dict):
raise ObservationValidationError("case must be an object")
_exact_keys(case, {"case_id", "description", "subject_id", "subject", "evidence", "expected_observations"}, "case")
for field in ("case_id", "description", "subject_id", "subject"):
_text(case[field], f"case.{field}")
evidence = case["evidence"]
if not isinstance(evidence, list) or not evidence:
raise ObservationValidationError("case.evidence must be a non-empty array")
seen: set[str] = set()
for index, unit in enumerate(evidence):
location = f"case.evidence[{index}]"
if not isinstance(unit, dict):
raise ObservationValidationError(f"{location} must be an object")
_exact_keys(unit, {"evidence_id", "text"}, location)
evidence_id = _text(unit["evidence_id"], f"{location}.evidence_id")
if evidence_id in seen:
raise ObservationValidationError(f"duplicate evidence ID: {evidence_id}")
seen.add(evidence_id)
_text(unit["text"], f"{location}.text")
expected = case["expected_observations"]
if not isinstance(expected, list) or not expected:
raise ObservationValidationError("case.expected_observations must be a non-empty array")
return case
def validate_fixture_case(case: dict[str, Any]) -> dict[str, Any]:
validate_case(case)
data = {"schema_version": SCHEMA_VERSION, "subject_id": case["subject_id"], "subject": case["subject"], "observations": case["expected_observations"]}
validate_observations(data, case)
return case
def build_prompt(case: dict[str, Any]) -> str:
validate_fixture_case(case)
model_input = {"subject_id": case["subject_id"], "subject": case["subject"], "evidence": case["evidence"]}
return PROMPT_TEMPLATE.format(input_json=json.dumps(model_input, ensure_ascii=False, indent=2))
def parse_model_json(raw_text: str) -> dict[str, Any]:
data = json.loads(raw_text)
if not isinstance(data, dict):
raise ObservationValidationError("model response JSON must be an object")
return data
def build_ollama_payload(model: str, prompt: str, num_ctx: int, num_predict: int) -> dict[str, Any]:
return {"model": model, "prompt": prompt, "think": False, "stream": False, "format": "json", "options": {"temperature": 0, "num_ctx": num_ctx, "num_predict": num_predict}}
def call_ollama(endpoint: str, model: str, prompt: str, timeout: int, num_ctx: int, num_predict: int) -> tuple[str, dict[str, Any]]:
started = time.perf_counter()
response = requests.post(endpoint, json=build_ollama_payload(model, prompt, num_ctx, num_predict), timeout=timeout)
elapsed = time.perf_counter() - started
response.raise_for_status()
body = response.json()
raw_text = body.get("response") if isinstance(body, dict) else None
if not isinstance(raw_text, str) or not raw_text.strip():
raise ValueError("Ollama returned no usable response text")
metadata = {
"model": body.get("model", model), "elapsed_seconds": round(elapsed, 3),
"total_duration_ns": body.get("total_duration"), "load_duration_ns": body.get("load_duration"),
"prompt_eval_count": body.get("prompt_eval_count"), "prompt_eval_duration_ns": body.get("prompt_eval_duration"),
"eval_count": body.get("eval_count"), "eval_duration_ns": body.get("eval_duration"),
"configuration": {"temperature": 0, "think": False, "num_ctx": num_ctx, "num_predict": num_predict, "retries": 0},
}
return raw_text.strip(), metadata
COMPARE_FIELDS = ("evidence_id", "target", "relation", "modality", "temporality", "evaluation", "agreement", "responsibility", "person", "uncertainty", "clarification_need")
def _scope_matches(actual: str, expected: str) -> bool:
if expected == "absent":
return actual == "absent"
expected_terms = [term.strip().casefold() for term in expected.split("|")]
folded = actual.casefold()
return any(term in folded for term in expected_terms)
def evaluate_observations(data: dict[str, Any], expected: list[dict[str, Any]]) -> dict[str, Any]:
actual = data["observations"]
checks: list[dict[str, Any]] = []
pair_count = min(len(actual), len(expected))
checks.append({"name": "observation_count", "passed": len(actual) == len(expected), "critical": False})
categories = {"missing_observations": max(0, len(expected) - len(actual)), "invented_observations": max(0, len(actual) - len(expected)), "stronger_commitment": 0, "weaker_commitment": 0, "incorrect_targets_relations": 0, "incorrect_responsibility": 0, "incorrect_uncertainty_clarification": 0}
commitment_rank = {"factual": 0, "possible": 1, "suggested": 1, "information_question": 1, "impersonal_necessity": 2, "interpersonal_request": 2, "committed": 3}
for index in range(pair_count):
got, want = actual[index], expected[index]
for field in COMPARE_FIELDS:
passed = got[field] == want[field]
checks.append({"name": f"obs_{index + 1}:{field}", "passed": passed, "critical": field in {"evidence_id", "target", "relation", "modality", "agreement", "responsibility", "person"}})
if not passed:
if field in {"target", "relation"}: categories["incorrect_targets_relations"] += 1
if field in {"responsibility", "person"}: categories["incorrect_responsibility"] += 1
if field in {"uncertainty", "clarification_need"}: categories["incorrect_uncertainty_clarification"] += 1
scope_ok = _scope_matches(got["scope"], want["scope"])
checks.append({"name": f"obs_{index + 1}:scope", "passed": scope_ok, "critical": False})
got_rank, want_rank = commitment_rank[got["modality"]], commitment_rank[want["modality"]]
if got_rank > want_rank or (want["agreement"] == "none" and got["agreement"] in {"accepted", "rejected"}): categories["stronger_commitment"] += 1
if got_rank < want_rank or (want["agreement"] in {"accepted", "rejected"} and got["agreement"] == "none"): categories["weaker_commitment"] += 1
passed_count = sum(check["passed"] for check in checks)
critical_failures = [check["name"] for check in checks if check["critical"] and not check["passed"]]
ratio = passed_count / len(checks)
if ratio == 1:
verdict = "PASS"
elif ratio >= 0.7 and categories["stronger_commitment"] == 0 and categories["incorrect_responsibility"] == 0:
verdict = "PARTIAL"
else:
verdict = "FAIL"
return {"verdict": verdict, "matched_checks": passed_count, "check_count": len(checks), "match_ratio": round(ratio, 3), "critical_failures": critical_failures, "error_categories": categories, "checks": checks}
def load_fixture(path: Path) -> list[dict[str, Any]]:
data = json.loads(path.read_text(encoding="utf-8-sig"))
if not isinstance(data, dict) or set(data) != {"cases"} or not isinstance(data["cases"], list) or not data["cases"]:
raise ObservationValidationError("fixture must contain exactly one non-empty cases list")
seen: set[str] = set()
for case in data["cases"]:
validate_fixture_case(case)
if case["case_id"] in seen:
raise ObservationValidationError(f"duplicate case ID: {case['case_id']}")
seen.add(case["case_id"])
return data["cases"]
def _write_json(path: Path, value: Any) -> None:
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
def run_case(case: dict[str, Any], output_root: Path, endpoint: str, model: str, timeout: int, num_ctx: int, num_predict: int) -> dict[str, Any]:
case_dir = output_root / case["case_id"]
case_dir.mkdir(parents=True, exist_ok=False)
_write_json(case_dir / "gold_input.json", {key: case[key] for key in ("case_id", "description", "subject_id", "subject", "evidence")})
_write_json(case_dir / "gold_expected_observations.json", case["expected_observations"])
prompt = build_prompt(case)
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
started = time.perf_counter()
try:
raw, metadata = call_ollama(endpoint, model, prompt, timeout, num_ctx, num_predict)
(case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8")
_write_json(case_dir / "ollama_metadata.json", metadata)
parsed = parse_model_json(raw)
_write_json(case_dir / "parsed_observations.json", parsed)
validate_observations(parsed, case)
validation = {"valid": True, "error": None}
evaluation = evaluate_observations(parsed, case["expected_observations"])
except requests.RequestException:
raise
except (json.JSONDecodeError, ObservationValidationError, ValueError) as exc:
validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)}
evaluation = {"verdict": "FAIL", "matched_checks": 0, "check_count": 0, "match_ratio": 0, "critical_failures": ["schema_validation"], "error_categories": {}, "checks": []}
_write_json(case_dir / "validation_result.json", validation)
result = {"case_id": case["case_id"], **evaluation, "elapsed_seconds": round(time.perf_counter() - started, 3)}
_write_json(case_dir / "evaluation_result.json", result)
return result
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
cases = load_fixture(args.fixture)
args.output.mkdir(parents=True, exist_ok=False)
started = time.perf_counter()
results = []
for index, case in enumerate(cases, start=1):
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
results.append(run_case(case, args.output, args.endpoint, args.model, args.timeout, args.num_ctx, args.num_predict))
summary = {"experiment": "evidence_near_observation_extraction", "schema_version": SCHEMA_VERSION, "model": args.model, "temperature": 0, "think": False, "retries": 0, "case_count": len(cases), "llm_call_count": len(results), "runtime_seconds": round(time.perf_counter() - started, 3), "verdict_counts": {v: sum(r["verdict"] == v for r in results) for v in ("PASS", "PARTIAL", "FAIL")}, "results": results}
_write_json(args.output / "summary.json", summary)
return summary
def main() -> int:
args = parse_args()
summary = run_experiment(args)
print(json.dumps(summary, ensure_ascii=False, indent=2))
return 0 if summary["verdict_counts"]["FAIL"] == 0 else 1
@@ -0,0 +1 @@
"""Reduced-semantic-load evidence observation experiment."""
@@ -0,0 +1,344 @@
#!/usr/bin/env python3
"""Extract reduced-semantic-load evidence-near observations."""
from __future__ import annotations
import argparse
import json
import re
import time
from pathlib import Path
from typing import Any
import requests
SCHEMA_VERSION = "experimental-evidence-observations-v2"
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
DEFAULT_MODEL = "qwen3.5:9B"
MODALITIES = {"factual", "possible", "suggested", "interpersonal_request", "impersonal_necessity", "information_question", "committed"}
TEMPORALITIES = {"existing", "future", "completed", "unspecified"}
EVALUATIONS = {"positive", "negative", "none"}
BINARY_SIGNALS = {"explicit", "absent"}
PRESENCE_SIGNALS = {"present", "absent"}
CLARIFICATION_NEEDS = {"explicit", "implicit", "none"}
OBSERVATION_ID_RE = re.compile(r"^obs_[1-9][0-9]*$")
class ObservationValidationError(ValueError):
"""Raised for invalid fixtures or model output."""
PROMPT_TEMPLATE = """You extract atomic linguistic and discourse observations for one fixed Discussion Subject.
Preserve only facts directly expressed by the evidence. Do not derive responsibility,
agreement, decisions, action items, open questions, accepted trials, rejected
alternatives, established actions, or protocol eligibility. Speaker identity, a name,
an addressee, first-person language, collective "we", and impersonal "man" never by
themselves establish responsibility.
Return exactly one JSON object with this shape:
{{
"schema_version": "experimental-evidence-observations-v2",
"subject_id": "copy exactly",
"subject": "copy exactly",
"observations": [
{{
"observation_id": "obs_1",
"evidence_id": "e1",
"content": "directly supported atomic observation",
"refers_to": null,
"speaker": "name copied from evidence or null",
"named_person": null,
"addressee": null,
"self_reference": false,
"collective_we": false,
"impersonal_person_reference": false,
"modality": "factual",
"temporality": "existing",
"evaluation": "none",
"affirmation": "absent",
"negation": "absent",
"determination_statement": "absent",
"uncertainty": "absent",
"clarification_need": "none",
"qualifier": null,
"limits_target": null
}}
]
}}
Rules:
- Produce multiple observations for distinct propositions in one evidence unit, but do
not fragment a single proposition unnecessarily.
- observation_id is sequential in evidence order. evidence_id must be copied exactly.
- refers_to is null or one earlier observation_id when the utterance explicitly refers
to it. Never use arrays. Preserve joint-reference utterances without inventing a
multi-target graph.
- speaker is the explicit transcript speaker. named_person is a person explicitly
named in the proposition. addressee is a person explicitly addressed.
- self_reference marks singular first-person self-reference. collective_we marks
collective first-person language. impersonal_person_reference marks impersonal
person expressions such as German "man".
- modality is factual, possible, suggested, interpersonal_request,
impersonal_necessity, information_question, or committed.
- temporality is existing, future, completed, or unspecified.
- evaluation is positive, negative, or none, only when linguistically supported.
- affirmation is explicit only for an explicit affirmative discourse signal such as
"ja". negation is explicit only for directly expressed negation/rejection.
- determination_statement is present only when the utterance explicitly says a
determination has been made.
- uncertainty is present or absent. clarification_need is explicit, implicit, or none.
- qualifier is null or concise evidence-grounded qualifying text.
- limits_target is null or one earlier observation explicitly limited in validity or
scope by this observation.
- Use JSON null, never the string "null". Output no fields beyond the schema.
Fixed Gold input:
{input_json}
"""
OBSERVATION_KEYS = {
"observation_id", "evidence_id", "content", "refers_to", "speaker",
"named_person", "addressee", "self_reference", "collective_we",
"impersonal_person_reference", "modality", "temporality", "evaluation",
"affirmation", "negation", "determination_statement", "uncertainty",
"clarification_need", "qualifier", "limits_target",
}
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser()
parser.add_argument("fixture", type=Path)
parser.add_argument("-o", "--output", type=Path, required=True)
parser.add_argument("--model", default=DEFAULT_MODEL)
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
parser.add_argument("--timeout", type=int, default=300)
parser.add_argument("--num-ctx", type=int, default=16384)
parser.add_argument("--num-predict", type=int, default=4096)
return parser.parse_args()
def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None:
missing, unknown = required - value.keys(), value.keys() - required
if missing:
raise ObservationValidationError(f"{location} missing required keys: {sorted(missing)}")
if unknown:
raise ObservationValidationError(f"{location} has unknown keys: {sorted(unknown)}")
def _text(value: Any, location: str) -> str:
if not isinstance(value, str) or not value.strip():
raise ObservationValidationError(f"{location} must be a non-empty string")
result = value.strip()
if result.casefold() == "null":
raise ObservationValidationError(f"{location} must not be the string 'null'")
return result
def _nullable_text(value: Any, location: str) -> None:
if value is not None:
_text(value, location)
def _prior_reference(value: Any, location: str, earlier: set[str]) -> None:
if value is None:
return
reference = _text(value, location)
if reference not in earlier:
raise ObservationValidationError(f"{location} references unknown or later observation: {reference}")
def validate_observations(data: Any, case: dict[str, Any]) -> dict[str, Any]:
validate_case(case)
if not isinstance(data, dict):
raise ObservationValidationError("output must be an object")
_exact_keys(data, {"schema_version", "subject_id", "subject", "observations"}, "output")
if data["schema_version"] != SCHEMA_VERSION:
raise ObservationValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
if data["subject_id"] != case["subject_id"] or data["subject"] != case["subject"]:
raise ObservationValidationError("model changed the fixed Discussion Subject")
observations = data["observations"]
if not isinstance(observations, list) or not observations:
raise ObservationValidationError("output.observations must be a non-empty array")
known_evidence = {item["evidence_id"] for item in case["evidence"]}
earlier: set[str] = set()
for index, observation in enumerate(observations, 1):
location = f"output.observations[{index - 1}]"
if not isinstance(observation, dict):
raise ObservationValidationError(f"{location} must be an object")
_exact_keys(observation, OBSERVATION_KEYS, location)
observation_id = _text(observation["observation_id"], f"{location}.observation_id")
if not OBSERVATION_ID_RE.fullmatch(observation_id) or observation_id != f"obs_{index}":
raise ObservationValidationError(f"{location}.observation_id must be obs_{index}")
evidence_id = _text(observation["evidence_id"], f"{location}.evidence_id")
if evidence_id not in known_evidence:
raise ObservationValidationError(f"{location}.evidence_id references unknown evidence: {evidence_id}")
_text(observation["content"], f"{location}.content")
_prior_reference(observation["refers_to"], f"{location}.refers_to", earlier)
_prior_reference(observation["limits_target"], f"{location}.limits_target", earlier)
for field in ("speaker", "named_person", "addressee", "qualifier"):
_nullable_text(observation[field], f"{location}.{field}")
for field in ("self_reference", "collective_we", "impersonal_person_reference"):
if not isinstance(observation[field], bool):
raise ObservationValidationError(f"{location}.{field} must be boolean")
for field, values in (
("modality", MODALITIES), ("temporality", TEMPORALITIES),
("evaluation", EVALUATIONS), ("affirmation", BINARY_SIGNALS),
("negation", BINARY_SIGNALS), ("determination_statement", PRESENCE_SIGNALS),
("uncertainty", PRESENCE_SIGNALS), ("clarification_need", CLARIFICATION_NEEDS),
):
if observation[field] not in values:
raise ObservationValidationError(f"{location}.{field} is invalid: {observation[field]!r}")
earlier.add(observation_id)
return data
def validate_case(case: Any) -> dict[str, Any]:
if not isinstance(case, dict):
raise ObservationValidationError("case must be an object")
_exact_keys(case, {"case_id", "description", "subject_id", "subject", "evidence", "expected_observations"}, "case")
for field in ("case_id", "description", "subject_id", "subject"):
_text(case[field], f"case.{field}")
if not isinstance(case["evidence"], list) or not case["evidence"]:
raise ObservationValidationError("case.evidence must be a non-empty array")
seen: set[str] = set()
for index, unit in enumerate(case["evidence"]):
_exact_keys(unit, {"evidence_id", "text"}, f"case.evidence[{index}]")
evidence_id = _text(unit["evidence_id"], f"case.evidence[{index}].evidence_id")
if evidence_id in seen:
raise ObservationValidationError(f"duplicate evidence ID: {evidence_id}")
seen.add(evidence_id)
_text(unit["text"], f"case.evidence[{index}].text")
if not isinstance(case["expected_observations"], list) or not case["expected_observations"]:
raise ObservationValidationError("case.expected_observations must be a non-empty array")
return case
def validate_fixture_case(case: dict[str, Any]) -> dict[str, Any]:
validate_case(case)
validate_observations({"schema_version": SCHEMA_VERSION, "subject_id": case["subject_id"], "subject": case["subject"], "observations": case["expected_observations"]}, case)
return case
def build_prompt(case: dict[str, Any]) -> str:
validate_fixture_case(case)
model_input = {key: case[key] for key in ("subject_id", "subject", "evidence")}
return PROMPT_TEMPLATE.format(input_json=json.dumps(model_input, ensure_ascii=False, indent=2))
def parse_model_json(raw_text: str) -> dict[str, Any]:
data = json.loads(raw_text)
if not isinstance(data, dict):
raise ObservationValidationError("model response JSON must be an object")
return data
def build_ollama_payload(model: str, prompt: str, num_ctx: int, num_predict: int) -> dict[str, Any]:
return {"model": model, "prompt": prompt, "think": False, "stream": False, "format": "json", "options": {"temperature": 0, "num_ctx": num_ctx, "num_predict": num_predict}}
def call_ollama(endpoint: str, model: str, prompt: str, timeout: int, num_ctx: int, num_predict: int) -> tuple[str, dict[str, Any]]:
started = time.perf_counter()
response = requests.post(endpoint, json=build_ollama_payload(model, prompt, num_ctx, num_predict), timeout=timeout)
elapsed = time.perf_counter() - started
response.raise_for_status()
body = response.json()
raw = body.get("response") if isinstance(body, dict) else None
if not isinstance(raw, str) or not raw.strip():
raise ValueError("Ollama returned no usable response text")
metadata = {"model": body.get("model", model), "elapsed_seconds": round(elapsed, 3), "total_duration_ns": body.get("total_duration"), "load_duration_ns": body.get("load_duration"), "prompt_eval_count": body.get("prompt_eval_count"), "prompt_eval_duration_ns": body.get("prompt_eval_duration"), "eval_count": body.get("eval_count"), "eval_duration_ns": body.get("eval_duration"), "configuration": {"temperature": 0, "think": False, "num_ctx": num_ctx, "num_predict": num_predict, "retries": 0}}
return raw.strip(), metadata
COMPARE_FIELDS = tuple(sorted(OBSERVATION_KEYS - {"observation_id", "content", "qualifier"}))
def _qualifier_matches(actual: str | None, expected: str | None) -> bool:
if expected is None:
return actual is None
if actual is None:
return False
return any(term.strip().casefold() in actual.casefold() for term in expected.split("|"))
def evaluate_observations(data: dict[str, Any], expected: list[dict[str, Any]]) -> dict[str, Any]:
actual = data["observations"]
checks = [{"name": "observation_count", "passed": len(actual) == len(expected), "critical": False}]
for index, (got, want) in enumerate(zip(actual, expected), 1):
for field in COMPARE_FIELDS:
checks.append({"name": f"obs_{index}:{field}", "passed": got[field] == want[field], "critical": field in {"evidence_id", "refers_to", "limits_target", "modality", "affirmation", "negation", "determination_statement"}})
checks.append({"name": f"obs_{index}:qualifier", "passed": _qualifier_matches(got["qualifier"], want["qualifier"]), "critical": False})
passed = sum(check["passed"] for check in checks)
ratio = passed / len(checks)
critical = [check["name"] for check in checks if check["critical"] and not check["passed"]]
verdict = "PASS" if ratio == 1 else "PARTIAL" if ratio >= 0.75 and not critical else "FAIL"
return {"verdict": verdict, "matched_checks": passed, "check_count": len(checks), "match_ratio": round(ratio, 3), "critical_failures": critical, "checks": checks}
def load_fixture(path: Path) -> list[dict[str, Any]]:
data = json.loads(path.read_text(encoding="utf-8-sig"))
if not isinstance(data, dict) or set(data) != {"cases"} or not isinstance(data["cases"], list) or not data["cases"]:
raise ObservationValidationError("fixture must contain exactly one non-empty cases list")
seen: set[str] = set()
for case in data["cases"]:
validate_fixture_case(case)
if case["case_id"] in seen:
raise ObservationValidationError(f"duplicate case ID: {case['case_id']}")
seen.add(case["case_id"])
return data["cases"]
def _write_json(path: Path, value: Any) -> None:
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
def run_case(case: dict[str, Any], output_root: Path, endpoint: str, model: str, timeout: int, num_ctx: int, num_predict: int) -> dict[str, Any]:
case_dir = output_root / case["case_id"]
case_dir.mkdir(parents=True, exist_ok=False)
_write_json(case_dir / "gold_input.json", {key: case[key] for key in ("case_id", "description", "subject_id", "subject", "evidence")})
_write_json(case_dir / "gold_expected_observations.json", case["expected_observations"])
prompt = build_prompt(case)
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
started = time.perf_counter()
raw, metadata = call_ollama(endpoint, model, prompt, timeout, num_ctx, num_predict)
(case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8")
_write_json(case_dir / "ollama_metadata.json", metadata)
try:
parsed = parse_model_json(raw)
_write_json(case_dir / "parsed_observations.json", parsed)
validate_observations(parsed, case)
validation = {"valid": True, "error": None}
evaluation = evaluate_observations(parsed, case["expected_observations"])
except (json.JSONDecodeError, ObservationValidationError, ValueError) as exc:
validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)}
evaluation = {"verdict": "FAIL", "matched_checks": 0, "check_count": 0, "match_ratio": 0, "critical_failures": ["schema_validation"], "checks": []}
_write_json(case_dir / "validation_result.json", validation)
result = {"case_id": case["case_id"], **evaluation, "elapsed_seconds": round(time.perf_counter() - started, 3)}
_write_json(case_dir / "evaluation_result.json", result)
return result
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
cases = load_fixture(args.fixture)
args.output.mkdir(parents=True, exist_ok=False)
started = time.perf_counter()
results = []
for index, case in enumerate(cases, 1):
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
results.append(run_case(case, args.output, args.endpoint, args.model, args.timeout, args.num_ctx, args.num_predict))
summary = {"experiment": "evidence_near_observation_extraction_v2", "schema_version": SCHEMA_VERSION, "model": args.model, "temperature": 0, "think": False, "retries": 0, "case_count": len(cases), "llm_call_count": len(results), "runtime_seconds": round(time.perf_counter() - started, 3), "verdict_counts": {verdict: sum(result["verdict"] == verdict for result in results) for verdict in ("PASS", "PARTIAL", "FAIL")}, "results": results}
_write_json(args.output / "summary.json", summary)
return summary
def main() -> int:
args = parse_args()
summary = run_experiment(args)
print(json.dumps(summary, ensure_ascii=False, indent=2))
return 0 if summary["verdict_counts"]["FAIL"] == 0 else 1
if __name__ == "__main__":
raise SystemExit(main())
@@ -0,0 +1 @@
"""Minimal semantic-preservation observation experiment."""
@@ -0,0 +1,271 @@
#!/usr/bin/env python3
"""Preserve meeting meaning as minimal atomic natural-language observations."""
from __future__ import annotations
import argparse
import json
import re
import time
from pathlib import Path
from typing import Any
import requests
SCHEMA_VERSION = "experimental-evidence-observations-v3"
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
DEFAULT_MODEL = "qwen3.5:9B"
OBSERVATION_ID_RE = re.compile(r"^obs_[1-9][0-9]*$")
OBSERVATION_KEYS = {"observation_id", "evidence_id", "content", "speaker", "named_person", "addressee"}
class ObservationValidationError(ValueError):
"""Raised for invalid fixtures or model output."""
PROMPT_TEMPLATE = """Preserve the meeting meaning in atomic natural-language observations.
This is semantic preservation, not classification or summarization. Return only facts
faithfully contributed by the evidence. Conservative wording is more important than
elegant prose. When in doubt, preserve the source wording closely.
Return exactly one JSON object:
{{
"schema_version": "experimental-evidence-observations-v3",
"subject_id": "copy exactly",
"subject": "copy exactly",
"observations": [
{{
"observation_id": "obs_1",
"evidence_id": "e1",
"content": "atomic, semantically faithful observation",
"speaker": "speaker copied from evidence",
"named_person": null,
"addressee": null
}}
]
}}
Rules:
- Use only the six observation fields shown. Do not output classifications, labels,
relations, scope fields, responsibility, agreement, decisions, actions, questions,
eligibility, or any other field.
- observation_id is sequential in evidence order. Copy evidence_id and speaker.
- named_person is null or a person explicitly named in that observation's evidence.
- addressee is null or a person explicitly addressed in that observation's evidence.
- A name, speaker, or addressee never implies responsibility, acceptance, ownership,
or assignment.
- content is not a summary. Preserve distinctions needed for later interpretation:
maybe/perhaps; can/could; should/must; personal, collective, or impersonal wording;
explicit requests, acceptances, and rejections; uncertainty and unresolved status;
conditions such as "if at all"; quantities; deadlines; trial/process/comparison
boundaries; "not yet"; and sequence such as "then".
- Never strengthen modality, weaken uncertainty, turn possibility into fact, turn a
preference into group rejection, turn a request into established work, turn "we"
into individual ownership, remove conditions/limits, generalize, or invent relations.
- Split one evidence unit only when it contributes propositions that may later require
different interpretations. Do not split merely because it has several clauses.
- Do not emit observation-ID relations. When evidence clearly makes an observation
depend on the immediately preceding proposition, state that dependency naturally in
content, without inventing an antecedent.
- Preserve content in the evidence language. Use JSON null, never the string "null".
Fixed input:
{input_json}
"""
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser()
parser.add_argument("fixture", type=Path)
parser.add_argument("-o", "--output", type=Path, required=True)
parser.add_argument("--model", default=DEFAULT_MODEL)
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
parser.add_argument("--timeout", type=int, default=300)
parser.add_argument("--num-ctx", type=int, default=16384)
parser.add_argument("--num-predict", type=int, default=4096)
return parser.parse_args()
def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None:
missing, unknown = required - value.keys(), value.keys() - required
if missing:
raise ObservationValidationError(f"{location} missing required keys: {sorted(missing)}")
if unknown:
raise ObservationValidationError(f"{location} has unknown keys: {sorted(unknown)}")
def _text(value: Any, location: str) -> str:
if not isinstance(value, str) or not value.strip():
raise ObservationValidationError(f"{location} must be a non-empty string")
result = value.strip()
if result.casefold() == "null":
raise ObservationValidationError(f"{location} must not be the string 'null'")
return result
def _explicit_people(text: str) -> set[str]:
prefix = text.split(":", 1)[0].strip() if ":" in text else ""
candidates = set(re.findall(r"\b(?:Dr\.\s+)?[A-ZÄÖÜ][A-Za-zÄÖÜäöüß-]+(?:\s+[A-ZÄÖÜ][A-Za-zÄÖÜäöüß-]+)*", text))
candidates.discard(prefix)
return candidates
def validate_observations(data: Any, case: dict[str, Any]) -> dict[str, Any]:
validate_case(case)
if not isinstance(data, dict):
raise ObservationValidationError("output must be an object")
_exact_keys(data, {"schema_version", "subject_id", "subject", "observations"}, "output")
if data["schema_version"] != SCHEMA_VERSION:
raise ObservationValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
if data["subject_id"] != case["subject_id"] or data["subject"] != case["subject"]:
raise ObservationValidationError("model changed the fixed Discussion Subject")
observations = data["observations"]
if not isinstance(observations, list) or not observations:
raise ObservationValidationError("output.observations must be a non-empty list")
evidence = {item["evidence_id"]: item["text"] for item in case["evidence"]}
seen: set[str] = set()
for index, observation in enumerate(observations):
location = f"output.observations[{index}]"
if not isinstance(observation, dict):
raise ObservationValidationError(f"{location} must be an object")
_exact_keys(observation, OBSERVATION_KEYS, location)
observation_id = _text(observation["observation_id"], f"{location}.observation_id")
if not OBSERVATION_ID_RE.fullmatch(observation_id) or observation_id in seen:
raise ObservationValidationError(f"{location}.observation_id must be unique and match obs_N")
seen.add(observation_id)
evidence_id = _text(observation["evidence_id"], f"{location}.evidence_id")
if evidence_id not in evidence:
raise ObservationValidationError(f"{location}.evidence_id references unknown evidence: {evidence_id}")
source = evidence[evidence_id]
source_speaker = source.split(":", 1)[0].strip()
speaker = _text(observation["speaker"], f"{location}.speaker")
if speaker != source_speaker:
raise ObservationValidationError(f"{location}.speaker must match evidence speaker {source_speaker!r}")
_text(observation["content"], f"{location}.content")
explicit_people = _explicit_people(source)
for field in ("named_person", "addressee"):
person = observation[field]
if person is not None:
person = _text(person, f"{location}.{field}")
if person not in explicit_people:
raise ObservationValidationError(f"{location}.{field} is not an explicit person in evidence: {person!r}")
return data
def validate_case(case: Any) -> dict[str, Any]:
required = {"case_id", "description", "subject_id", "subject", "evidence", "semantic_requirements"}
if not isinstance(case, dict):
raise ObservationValidationError("case must be an object")
_exact_keys(case, required, "case")
for field in ("case_id", "description", "subject_id", "subject"):
_text(case[field], f"case.{field}")
if not isinstance(case["evidence"], list) or not case["evidence"]:
raise ObservationValidationError("case.evidence must be a non-empty list")
evidence_ids: set[str] = set()
for index, unit in enumerate(case["evidence"]):
_exact_keys(unit, {"evidence_id", "text"}, f"case.evidence[{index}]")
evidence_id = _text(unit["evidence_id"], f"case.evidence[{index}].evidence_id")
if evidence_id in evidence_ids:
raise ObservationValidationError(f"duplicate evidence ID: {evidence_id}")
evidence_ids.add(evidence_id)
_text(unit["text"], f"case.evidence[{index}].text")
if not isinstance(case["semantic_requirements"], list) or not case["semantic_requirements"]:
raise ObservationValidationError("case.semantic_requirements must be a non-empty list")
for index, requirement in enumerate(case["semantic_requirements"]):
_text(requirement, f"case.semantic_requirements[{index}]")
return case
def build_prompt(case: dict[str, Any]) -> str:
validate_case(case)
model_input = {key: case[key] for key in ("subject_id", "subject", "evidence")}
return PROMPT_TEMPLATE.format(input_json=json.dumps(model_input, ensure_ascii=False, indent=2))
def parse_model_json(raw_text: str) -> dict[str, Any]:
data = json.loads(raw_text)
if not isinstance(data, dict):
raise ObservationValidationError("model response JSON must be an object")
return data
def build_ollama_payload(model: str, prompt: str, num_ctx: int, num_predict: int) -> dict[str, Any]:
return {"model": model, "prompt": prompt, "think": False, "stream": False, "format": "json", "options": {"temperature": 0, "num_ctx": num_ctx, "num_predict": num_predict}}
def call_ollama(endpoint: str, model: str, prompt: str, timeout: int, num_ctx: int, num_predict: int) -> tuple[str, dict[str, Any]]:
started = time.perf_counter()
response = requests.post(endpoint, json=build_ollama_payload(model, prompt, num_ctx, num_predict), timeout=timeout)
elapsed = time.perf_counter() - started
response.raise_for_status()
body = response.json()
raw = body.get("response") if isinstance(body, dict) else None
if not isinstance(raw, str) or not raw.strip():
raise ValueError("Ollama returned no usable response text")
metadata = {"model": body.get("model", model), "elapsed_seconds": round(elapsed, 3), "total_duration_ns": body.get("total_duration"), "load_duration_ns": body.get("load_duration"), "prompt_eval_count": body.get("prompt_eval_count"), "prompt_eval_duration_ns": body.get("prompt_eval_duration"), "eval_count": body.get("eval_count"), "eval_duration_ns": body.get("eval_duration"), "configuration": {"temperature": 0, "think": False, "num_ctx": num_ctx, "num_predict": num_predict, "retries": 0}}
return raw.strip(), metadata
def load_fixture(path: Path) -> list[dict[str, Any]]:
data = json.loads(path.read_text(encoding="utf-8-sig"))
if not isinstance(data, dict) or set(data) != {"cases"} or not isinstance(data["cases"], list) or not data["cases"]:
raise ObservationValidationError("fixture must contain exactly one non-empty cases list")
seen: set[str] = set()
for case in data["cases"]:
validate_case(case)
if case["case_id"] in seen:
raise ObservationValidationError(f"duplicate case ID: {case['case_id']}")
seen.add(case["case_id"])
return data["cases"]
def _write_json(path: Path, value: Any) -> None:
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
def run_case(case: dict[str, Any], output_root: Path, endpoint: str, model: str, timeout: int, num_ctx: int, num_predict: int) -> dict[str, Any]:
case_dir = output_root / case["case_id"]
case_dir.mkdir(parents=True, exist_ok=False)
_write_json(case_dir / "source_evidence.json", {key: case[key] for key in ("case_id", "description", "subject_id", "subject", "evidence")})
_write_json(case_dir / "gold_semantic_requirements.json", case["semantic_requirements"])
prompt = build_prompt(case)
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
started = time.perf_counter()
raw, metadata = call_ollama(endpoint, model, prompt, timeout, num_ctx, num_predict)
(case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8")
_write_json(case_dir / "ollama_metadata.json", metadata)
try:
parsed = parse_model_json(raw)
_write_json(case_dir / "parsed_observations.json", parsed)
validate_observations(parsed, case)
validation = {"valid": True, "error": None}
except (json.JSONDecodeError, ObservationValidationError, ValueError) as exc:
validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)}
_write_json(case_dir / "structural_validation.json", validation)
return {"case_id": case["case_id"], "structurally_valid": validation["valid"], "elapsed_seconds": round(time.perf_counter() - started, 3)}
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
cases = load_fixture(args.fixture)
args.output.mkdir(parents=True, exist_ok=False)
started = time.perf_counter()
results = []
for index, case in enumerate(cases, 1):
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
results.append(run_case(case, args.output, args.endpoint, args.model, args.timeout, args.num_ctx, args.num_predict))
summary = {"experiment": "evidence_near_observation_extraction_v3", "schema_version": SCHEMA_VERSION, "model": args.model, "temperature": 0, "think": False, "retries": 0, "case_count": len(cases), "llm_call_count": len(results), "runtime_seconds": round(time.perf_counter() - started, 3), "structurally_valid_count": sum(result["structurally_valid"] for result in results), "results": results}
_write_json(args.output / "summary.json", summary)
return summary
def main() -> int:
args = parse_args()
summary = run_experiment(args)
print(json.dumps(summary, ensure_ascii=False, indent=2))
return 0 if summary["structurally_valid_count"] == summary["case_count"] else 1
if __name__ == "__main__":
raise SystemExit(main())
@@ -0,0 +1 @@
"""Isolated experimental semantic synthesis for known discussion subjects."""
@@ -0,0 +1,640 @@
#!/usr/bin/env python3
"""Run semantic synthesis with subject detection and evidence assignment fixed."""
from __future__ import annotations
import argparse
import json
import time
from pathlib import Path
from typing import Any
import requests
SCHEMA_VERSION = "experimental-semantic-synthesis-v1"
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
DEFAULT_MODEL = "qwen3.5:9B"
DEFAULT_TIMEOUT = 300
DEFAULT_NUM_CTX = 8192
DEFAULT_NUM_PREDICT = 2048
EVENT_TYPES = {
"idea",
"option",
"proposal",
"objection",
"supporting_argument",
"clarification",
"rejection",
"scoped_acceptance",
"fact",
"technical_finding",
}
OUTCOME_STATUSES = {"established", "rejected", "scoped_acceptance", "tentative"}
class SynthesisValidationError(ValueError):
"""Raised when isolated semantic synthesis output is structurally invalid."""
PROMPT_TEMPLATE = """You perform semantic synthesis for one already known discussion subject.
The subject boundary and evidence assignment are fixed and complete. Do not discover,
split, merge, rename, or omit the subject. Do not assign evidence to another subject.
Interpret only what the supplied evidence semantically establishes.
Semantic distinctions:
- idea: mentioned possibility without stronger commitment
- option: alternative considered without commitment
- proposal: suggested course of action not yet established as work
- objection: argument or concern against something; not automatically unresolved
- rejection: an alternative is explicitly rejected
- scoped_acceptance: accepted only for the stated test, trial, condition, or scope
- proposal is not an action
- no decision is not a tentative decision
- mention is not an unresolved issue
- an action requires explicit assignment, acceptance, commitment, or established work
- an unresolved issue requires a concrete need explicitly left unresolved
Preserve explicit rejection, explicit accepted work, explicit unresolved questions,
and all limits on an outcome. Never generalize trial acceptance into final acceptance.
Use only supplied evidence IDs. Keep concise semantic text in the evidence language.
Return exactly one JSON object. Always include these fields:
{{
"schema_version": "experimental-semantic-synthesis-v1",
"subject_id": "copy the supplied subject_id exactly",
"subject": "copy the supplied subject exactly",
"events": [
{{
"type": "idea|option|proposal|objection|supporting_argument|clarification|rejection|scoped_acceptance|fact|technical_finding",
"text": "supported semantic event",
"evidence_ids": ["e1"]
}}
],
"actions": [
{{
"text": "established action",
"responsible": null,
"due": null,
"evidence_ids": ["e2"]
}}
],
"unresolved_issues": [
{{
"text": "explicitly unresolved issue",
"evidence_ids": ["e3"]
}}
]
}}
The three arrays are structurally required; use [] when none exist.
Add "outcome" only when an outcome was actually established:
{{
"status": "established|rejected|scoped_acceptance|tentative",
"text": "what was actually established",
"scope": "the exact scope, condition, or limit",
"evidence_ids": ["e2"]
}}
Omit outcome completely when there is none. Never use null for outcome. Never use the
string "null"; use JSON null only for unknown responsible or due values.
Fixed Gold input:
{input_json}
"""
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Run the isolated semantic-synthesis Gold experiment."
)
parser.add_argument("fixture", type=Path, help="Fixed-subject Gold bundle JSON.")
parser.add_argument("-o", "--output", type=Path, required=True)
parser.add_argument("--model", default=DEFAULT_MODEL)
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT)
parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX)
parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT)
return parser.parse_args()
def _exact_keys(
value: dict[str, Any], required: set[str], optional: set[str], location: str
) -> None:
missing = required - value.keys()
unknown = value.keys() - required - optional
if missing:
raise SynthesisValidationError(
f"{location} missing required keys: {sorted(missing)}"
)
if unknown:
raise SynthesisValidationError(
f"{location} has unknown keys: {sorted(unknown)}"
)
def _text(value: Any, location: str) -> str:
if not isinstance(value, str) or not value.strip():
raise SynthesisValidationError(f"{location} must be a non-empty string")
return value.strip()
def validate_bundle(case: Any) -> dict[str, Any]:
if not isinstance(case, dict):
raise SynthesisValidationError("case must be an object")
_exact_keys(
case,
{
"case_id",
"description",
"subject_id",
"subject",
"evidence",
"allowed_responsible",
"expected",
},
set(),
"case",
)
_text(case["case_id"], "case.case_id")
_text(case["description"], "case.description")
_text(case["subject_id"], "case.subject_id")
_text(case["subject"], "case.subject")
evidence = case["evidence"]
if not isinstance(evidence, list) or not evidence:
raise SynthesisValidationError("case.evidence must be a non-empty list")
seen: set[str] = set()
for index, item in enumerate(evidence):
location = f"case.evidence[{index}]"
if not isinstance(item, dict):
raise SynthesisValidationError(f"{location} must be an object")
_exact_keys(item, {"evidence_id", "text"}, set(), location)
evidence_id = _text(item["evidence_id"], f"{location}.evidence_id")
if evidence_id in seen:
raise SynthesisValidationError(f"duplicate evidence ID: {evidence_id}")
seen.add(evidence_id)
_text(item["text"], f"{location}.text")
allowed = case["allowed_responsible"]
if not isinstance(allowed, list) or any(
not isinstance(value, str) or not value.strip() for value in allowed
):
raise SynthesisValidationError(
"case.allowed_responsible must be a list of non-empty strings"
)
if len(set(allowed)) != len(allowed):
raise SynthesisValidationError("case.allowed_responsible contains duplicates")
if not isinstance(case["expected"], dict):
raise SynthesisValidationError("case.expected must be an object")
return case
def _evidence_ids(value: Any, location: str, known: set[str]) -> list[str]:
if not isinstance(value, list) or not value:
raise SynthesisValidationError(f"{location} must be a non-empty list")
result: list[str] = []
for index, evidence_id in enumerate(value):
evidence_id = _text(evidence_id, f"{location}[{index}]")
if evidence_id not in known:
raise SynthesisValidationError(
f"{location}[{index}] references unknown evidence ID: {evidence_id}"
)
if evidence_id in result:
raise SynthesisValidationError(
f"{location} contains duplicate evidence ID: {evidence_id}"
)
result.append(evidence_id)
return result
def _nullable_text(value: Any, location: str) -> str | None:
if value is None:
return None
result = _text(value, location)
if result.casefold() == "null":
raise SynthesisValidationError(
f"{location} must use JSON null, not the string 'null'"
)
return result
def validate_synthesis(data: Any, case: dict[str, Any]) -> dict[str, Any]:
validate_bundle(case)
if not isinstance(data, dict):
raise SynthesisValidationError("output must be an object")
_exact_keys(
data,
{
"schema_version",
"subject_id",
"subject",
"events",
"actions",
"unresolved_issues",
},
{"outcome"},
"output",
)
if data["schema_version"] != SCHEMA_VERSION:
raise SynthesisValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
if data["subject_id"] != case["subject_id"]:
raise SynthesisValidationError("model changed fixed subject_id")
if data["subject"] != case["subject"]:
raise SynthesisValidationError("model changed fixed subject")
known = {item["evidence_id"] for item in case["evidence"]}
events = data["events"]
if not isinstance(events, list):
raise SynthesisValidationError("output.events must be an array")
for index, event in enumerate(events):
location = f"output.events[{index}]"
if not isinstance(event, dict):
raise SynthesisValidationError(f"{location} must be an object")
_exact_keys(event, {"type", "text", "evidence_ids"}, set(), location)
if event["type"] not in EVENT_TYPES:
raise SynthesisValidationError(f"{location}.type is invalid")
_text(event["text"], f"{location}.text")
_evidence_ids(event["evidence_ids"], f"{location}.evidence_ids", known)
if "outcome" in data:
outcome = data["outcome"]
if not isinstance(outcome, dict):
raise SynthesisValidationError(
"output.outcome must be an object when present; omit it when absent"
)
_exact_keys(
outcome, {"status", "text", "scope", "evidence_ids"}, set(), "output.outcome"
)
if outcome["status"] not in OUTCOME_STATUSES:
raise SynthesisValidationError("output.outcome.status is invalid")
_text(outcome["text"], "output.outcome.text")
_text(outcome["scope"], "output.outcome.scope")
_evidence_ids(outcome["evidence_ids"], "output.outcome.evidence_ids", known)
actions = data["actions"]
if not isinstance(actions, list):
raise SynthesisValidationError("output.actions must be an array")
allowed = set(case["allowed_responsible"])
for index, action in enumerate(actions):
location = f"output.actions[{index}]"
if not isinstance(action, dict):
raise SynthesisValidationError(f"{location} must be an object")
_exact_keys(
action,
{"text", "responsible", "due", "evidence_ids"},
set(),
location,
)
_text(action["text"], f"{location}.text")
responsible = _nullable_text(action["responsible"], f"{location}.responsible")
if responsible is not None and responsible not in allowed:
raise SynthesisValidationError(
f"{location}.responsible is not allowed: {responsible}"
)
_nullable_text(action["due"], f"{location}.due")
_evidence_ids(action["evidence_ids"], f"{location}.evidence_ids", known)
issues = data["unresolved_issues"]
if not isinstance(issues, list):
raise SynthesisValidationError("output.unresolved_issues must be an array")
for index, issue in enumerate(issues):
location = f"output.unresolved_issues[{index}]"
if not isinstance(issue, dict):
raise SynthesisValidationError(f"{location} must be an object")
_exact_keys(issue, {"text", "evidence_ids"}, set(), location)
_text(issue["text"], f"{location}.text")
_evidence_ids(issue["evidence_ids"], f"{location}.evidence_ids", known)
return data
def build_prompt(case: dict[str, Any]) -> str:
validate_bundle(case)
model_input = {
"subject_id": case["subject_id"],
"subject": case["subject"],
"evidence": case["evidence"],
}
return PROMPT_TEMPLATE.format(
input_json=json.dumps(model_input, ensure_ascii=False, indent=2)
)
def parse_model_json(raw_text: str) -> dict[str, Any]:
data = json.loads(raw_text)
if not isinstance(data, dict):
raise SynthesisValidationError("model response JSON must be an object")
return data
def build_ollama_payload(
model: str, prompt: str, num_ctx: int, num_predict: int
) -> dict[str, Any]:
return {
"model": model,
"prompt": prompt,
"think": False,
"stream": False,
"format": "json",
"options": {
"temperature": 0,
"num_ctx": num_ctx,
"num_predict": num_predict,
},
}
def call_ollama(
endpoint: str,
model: str,
prompt: str,
timeout: int,
num_ctx: int,
num_predict: int,
) -> tuple[str, dict[str, Any]]:
payload = build_ollama_payload(model, prompt, num_ctx, num_predict)
started = time.perf_counter()
response = requests.post(endpoint, json=payload, timeout=timeout)
elapsed = time.perf_counter() - started
response.raise_for_status()
body = response.json()
if not isinstance(body, dict):
raise ValueError("Ollama response must be an object")
raw_text = body.get("response")
if not isinstance(raw_text, str) or not raw_text.strip():
raise ValueError("Ollama returned no usable response text")
metadata = {
"model": body.get("model", model),
"elapsed_seconds": round(elapsed, 3),
"total_duration_ns": body.get("total_duration"),
"load_duration_ns": body.get("load_duration"),
"prompt_eval_count": body.get("prompt_eval_count"),
"prompt_eval_duration_ns": body.get("prompt_eval_duration"),
"eval_count": body.get("eval_count"),
"eval_duration_ns": body.get("eval_duration"),
"configuration": {
"temperature": 0,
"think": False,
"num_ctx": num_ctx,
"num_predict": num_predict,
},
}
return raw_text.strip(), metadata
def _contains(text: str, terms: list[str]) -> bool:
folded = text.casefold()
return any(term.casefold() in folded for term in terms)
def _refs_cover(items: list[dict[str, Any]], expected: list[str]) -> bool:
actual = {
evidence_id
for item in items
for evidence_id in item.get("evidence_ids", [])
}
return set(expected).issubset(actual)
def evaluate_synthesis(data: dict[str, Any], expected: dict[str, Any]) -> dict[str, Any]:
checks: list[dict[str, Any]] = []
def add(name: str, passed: bool, critical: bool = False) -> None:
checks.append({"name": name, "passed": passed, "critical": critical})
events = data["events"]
event_types = [item["type"] for item in events]
for event_type, minimum in expected.get("event_type_minimums", {}).items():
add(f"event:{event_type}", event_types.count(event_type) >= minimum)
allowed_types = set(expected.get("allowed_event_types", EVENT_TYPES))
add("no_unexpected_event_types", set(event_types).issubset(allowed_types))
add(
"event_evidence",
_refs_cover(events, expected.get("event_evidence_ids", [])),
)
outcome_expected = expected["outcome"]
outcome = data.get("outcome")
add(
"outcome_presence",
(outcome is not None) == outcome_expected["required"],
critical=True,
)
if outcome_expected["required"] and outcome is not None:
add("outcome_status", outcome["status"] in outcome_expected["statuses"])
combined = f"{outcome['text']} {outcome['scope']}"
add("outcome_meaning", _contains(combined, outcome_expected["terms"]))
add(
"outcome_scope",
_contains(combined, outcome_expected["scope_terms"]),
critical=True,
)
add(
"outcome_evidence",
set(outcome_expected["evidence_ids"]).issubset(outcome["evidence_ids"]),
critical=True,
)
actions = data["actions"]
expected_actions = expected["actions"]
add(
"action_count",
len(actions) == expected_actions["count"],
critical=True,
)
if expected_actions["count"] and actions:
action_text = " ".join(item["text"] for item in actions)
add("action_meaning", _contains(action_text, expected_actions["terms"]))
if "responsible" in expected_actions:
add(
"action_responsible",
any(item["responsible"] == expected_actions["responsible"] for item in actions),
critical=True,
)
if expected_actions.get("due_terms"):
due_text = " ".join(str(item["due"] or "") for item in actions)
add("action_due", _contains(due_text, expected_actions["due_terms"]))
add(
"action_evidence",
_refs_cover(actions, expected_actions["evidence_ids"]),
critical=True,
)
issues = data["unresolved_issues"]
expected_issues = expected["unresolved_issues"]
add(
"unresolved_count",
len(issues) == expected_issues["count"],
critical=True,
)
if expected_issues["count"] and issues:
issue_text = " ".join(item["text"] for item in issues)
add("unresolved_meaning", _contains(issue_text, expected_issues["terms"]))
add(
"unresolved_evidence",
_refs_cover(issues, expected_issues["evidence_ids"]),
critical=True,
)
passed = sum(item["passed"] for item in checks)
critical_failures = [
item["name"] for item in checks if item["critical"] and not item["passed"]
]
ratio = passed / len(checks)
if ratio == 1:
verdict = "PASS"
elif ratio >= 0.7 and not critical_failures:
verdict = "PARTIAL"
else:
verdict = "FAIL"
failed = [item["name"] for item in checks if not item["passed"]]
return {
"verdict": verdict,
"reason": "All semantic checks passed." if not failed else "Failed: " + ", ".join(failed),
"passed_checks": passed,
"check_count": len(checks),
"critical_failures": critical_failures,
"checks": checks,
}
def load_fixture(path: Path) -> list[dict[str, Any]]:
data = json.loads(path.read_text(encoding="utf-8-sig"))
if not isinstance(data, dict) or set(data) != {"cases"}:
raise SynthesisValidationError("fixture must contain exactly a cases list")
cases = data["cases"]
if not isinstance(cases, list) or not cases:
raise SynthesisValidationError("fixture cases must be a non-empty list")
seen: set[str] = set()
for case in cases:
validate_bundle(case)
if case["case_id"] in seen:
raise SynthesisValidationError(f"duplicate case ID: {case['case_id']}")
seen.add(case["case_id"])
return cases
def run_case(
case: dict[str, Any],
output_root: Path,
endpoint: str,
model: str,
timeout: int,
num_ctx: int,
num_predict: int,
) -> dict[str, Any]:
case_dir = output_root / case["case_id"]
case_dir.mkdir(parents=True, exist_ok=False)
gold_input = {
"case_id": case["case_id"],
"description": case["description"],
"subject_id": case["subject_id"],
"subject": case["subject"],
"evidence": case["evidence"],
}
(case_dir / "gold_input.json").write_text(
json.dumps(gold_input, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
prompt = build_prompt(case)
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
started = time.perf_counter()
try:
raw_text, metadata = call_ollama(
endpoint, model, prompt, timeout, num_ctx, num_predict
)
(case_dir / "raw_model_response.txt").write_text(raw_text + "\n", encoding="utf-8")
(case_dir / "ollama_metadata.json").write_text(
json.dumps(metadata, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
parsed = parse_model_json(raw_text)
(case_dir / "parsed_response.json").write_text(
json.dumps(parsed, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
validated = validate_synthesis(parsed, case)
validation = {"valid": True, "error": None}
evaluation = evaluate_synthesis(validated, case["expected"])
except requests.RequestException:
raise
except (json.JSONDecodeError, SynthesisValidationError, ValueError) as exc:
validation = {
"valid": False,
"error_type": type(exc).__name__,
"error": str(exc),
}
evaluation = {
"verdict": "FAIL",
"reason": f"Schema validation failed: {exc}",
"passed_checks": 0,
"check_count": 0,
"critical_failures": ["schema_validation"],
"checks": [],
}
(case_dir / "validation_result.json").write_text(
json.dumps(validation, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
result = {
"case_id": case["case_id"],
"description": case["description"],
**evaluation,
"elapsed_seconds": round(time.perf_counter() - started, 3),
}
(case_dir / "evaluation_result.json").write_text(
json.dumps(result, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
return result
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
cases = load_fixture(args.fixture)
args.output.mkdir(parents=True, exist_ok=False)
started = time.perf_counter()
results: list[dict[str, Any]] = []
for index, case in enumerate(cases, start=1):
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
results.append(
run_case(
case,
args.output,
args.endpoint,
args.model,
args.timeout,
args.num_ctx,
args.num_predict,
)
)
summary = {
"experiment": "semantic_synthesis_isolation",
"schema_version": SCHEMA_VERSION,
"model": args.model,
"temperature": 0,
"think": False,
"num_ctx": args.num_ctx,
"num_predict": args.num_predict,
"case_count": len(cases),
"llm_call_count": len(results),
"runtime_seconds": round(time.perf_counter() - started, 3),
"verdict_counts": {
verdict: sum(result["verdict"] == verdict for result in results)
for verdict in ("PASS", "PARTIAL", "FAIL")
},
"results": results,
}
(args.output / "summary.json").write_text(
json.dumps(summary, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
return summary
def main() -> int:
args = parse_args()
try:
summary = run_experiment(args)
except (OSError, ValueError, requests.RequestException) as exc:
print(f"Error: {exc}")
return 1
print(json.dumps(summary["verdict_counts"], sort_keys=True))
print(f"Artifacts: {args.output.resolve()}")
return 0
if __name__ == "__main__":
raise SystemExit(main())
@@ -0,0 +1 @@
"""Experimental topic-oriented discussion reconstruction."""
@@ -0,0 +1,721 @@
#!/usr/bin/env python3
"""Run an isolated Discussion Subject reconstruction experiment with Ollama."""
from __future__ import annotations
import argparse
import json
import re
import time
from pathlib import Path
from typing import Any
import requests
SCHEMA_VERSION = "experimental-discussion-subjects-v1"
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
DEFAULT_MODEL = "qwen3.5:9B"
DEFAULT_TIMEOUT = 300
DEFAULT_NUM_CTX = 16384
DEFAULT_NUM_PREDICT = 4096
EVENT_TYPES = {
"introduced_idea",
"considered_option",
"proposal",
"supporting_argument",
"objection",
"clarification",
"modification",
"fact",
"technical_finding",
}
OUTCOME_CERTAINTIES = {"established", "tentative", "conditional", "rejected"}
IDENTIFIER_RE = re.compile(r"^[a-z][a-z0-9_]*$")
class ReconstructionValidationError(ValueError):
"""Raised when experimental reconstruction output violates the schema."""
PROMPT_TEMPLATE = """You reconstruct discussion subjects from meeting evidence.
This is semantic reconstruction, not protocol writing and not flat category extraction.
Group evidence by what participants are actually discussing. For each subject, record
only supported discourse events and, when present, the actual outcome, resulting
actions, and genuinely unresolved issues.
Important distinctions:
- discussed is not necessarily proposed
- proposed is not necessarily preferred or accepted
- preferred is not accepted
- accepted for a trial is not accepted as a final solution
- mentioned is not an unresolved question
- an outcome must preserve its scope, conditions, polarity, and uncertainty
- do not infer responsibility from mention, expertise, adjacency, or likely role
- do not invent missing stages or emit empty optional structures
Evidence discipline:
- Use only the supplied evidence IDs in evidence_refs.
- Every subject, event, outcome, action, and unresolved issue needs at least one
evidence reference.
- Keep statements concise; do not copy long evidence passages.
- A subject may consist only of one introduced idea.
Return one JSON object with exactly:
{{
"schema_version": "experimental-discussion-subjects-v1",
"subjects": [
{{
"subject_id": "subject_1",
"title": "concise discussion subject",
"evidence_refs": ["e1"],
"development": [
{{
"event_id": "event_1",
"type": "introduced_idea|considered_option|proposal|supporting_argument|objection|clarification|modification|fact|technical_finding",
"text": "what happened in the discussion",
"evidence_refs": ["e1"]
}}
],
"outcome": {{
"text": "only what was established",
"scope": "explicit limit or full scope of the outcome",
"certainty": "established|tentative|conditional|rejected",
"evidence_refs": ["e2"]
}},
"actions": [
{{
"action_id": "action_1",
"text": "established work only",
"responsible": "explicitly supported name or null",
"deadline": "explicitly supported deadline or null",
"evidence_refs": ["e3"]
}}
],
"unresolved_issues": [
{{
"issue_id": "issue_1",
"text": "concrete unresolved issue",
"evidence_refs": ["e4"]
}}
]
}}
]
}}
Only subject_id, title, evidence_refs are required for each subject. Omit
development, outcome, actions, or unresolved_issues when absent. Never emit null
or an empty optional list/object.
Case ID: {case_id}
Evidence units:
{evidence_json}
"""
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Run the isolated topic-reconstruction Gold experiment."
)
parser.add_argument("fixture", type=Path, help="Focused Gold cases JSON.")
parser.add_argument("-o", "--output", type=Path, required=True)
parser.add_argument("--model", default=DEFAULT_MODEL)
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT)
parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX)
parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT)
parser.add_argument(
"--case", action="append", dest="case_ids", help="Run only this case ID."
)
return parser.parse_args()
def _expect_exact_keys(
value: dict[str, Any], required: set[str], optional: set[str], location: str
) -> None:
missing = required - value.keys()
unknown = value.keys() - required - optional
if missing:
raise ReconstructionValidationError(
f"{location} missing required keys: {sorted(missing)}"
)
if unknown:
raise ReconstructionValidationError(
f"{location} has unknown keys: {sorted(unknown)}"
)
def _nonempty_text(value: Any, location: str) -> str:
if not isinstance(value, str) or not value.strip():
raise ReconstructionValidationError(f"{location} must be a non-empty string")
return value.strip()
def _identifier(value: Any, location: str, seen: set[str]) -> str:
text = _nonempty_text(value, location)
if not IDENTIFIER_RE.fullmatch(text):
raise ReconstructionValidationError(f"{location} is not a valid identifier")
if text in seen:
raise ReconstructionValidationError(f"duplicate identifier: {text}")
seen.add(text)
return text
def _nullable_text(value: Any, location: str) -> str | None:
if value is None:
return None
text = _nonempty_text(value, location)
if text.casefold() == "null":
raise ReconstructionValidationError(
f"{location} must use JSON null, not the string 'null'"
)
return text
def _evidence_refs(value: Any, location: str, known: set[str]) -> list[str]:
if not isinstance(value, list) or not value:
raise ReconstructionValidationError(f"{location} must be a non-empty list")
refs: list[str] = []
for index, ref in enumerate(value):
ref = _nonempty_text(ref, f"{location}[{index}]")
if ref not in known:
raise ReconstructionValidationError(
f"{location}[{index}] references unknown evidence ID: {ref}"
)
if ref in refs:
raise ReconstructionValidationError(
f"{location} contains duplicate evidence reference: {ref}"
)
refs.append(ref)
return refs
def validate_evidence_units(evidence_units: Any) -> set[str]:
if not isinstance(evidence_units, list) or not evidence_units:
raise ReconstructionValidationError("evidence_units must be a non-empty list")
known: set[str] = set()
for index, unit in enumerate(evidence_units):
location = f"evidence_units[{index}]"
if not isinstance(unit, dict):
raise ReconstructionValidationError(f"{location} must be an object")
_expect_exact_keys(unit, {"evidence_id", "text"}, set(), location)
evidence_id = _nonempty_text(unit["evidence_id"], f"{location}.evidence_id")
if evidence_id in known:
raise ReconstructionValidationError(
f"duplicate input evidence identifier: {evidence_id}"
)
known.add(evidence_id)
_nonempty_text(unit["text"], f"{location}.text")
return known
def validate_reconstruction(data: Any, evidence_units: Any) -> dict[str, Any]:
known = validate_evidence_units(evidence_units)
if not isinstance(data, dict):
raise ReconstructionValidationError("model output must be an object")
_expect_exact_keys(data, {"schema_version", "subjects"}, set(), "output")
if data["schema_version"] != SCHEMA_VERSION:
raise ReconstructionValidationError(
f"schema_version must be {SCHEMA_VERSION!r}"
)
subjects = data["subjects"]
if not isinstance(subjects, list) or not subjects:
raise ReconstructionValidationError("subjects must be a non-empty list")
seen: set[str] = set()
for subject_index, subject in enumerate(subjects):
location = f"subjects[{subject_index}]"
if not isinstance(subject, dict):
raise ReconstructionValidationError(f"{location} must be an object")
_expect_exact_keys(
subject,
{"subject_id", "title", "evidence_refs"},
{"development", "outcome", "actions", "unresolved_issues"},
location,
)
_identifier(subject["subject_id"], f"{location}.subject_id", seen)
_nonempty_text(subject["title"], f"{location}.title")
_evidence_refs(subject["evidence_refs"], f"{location}.evidence_refs", known)
if "development" in subject:
events = subject["development"]
if not isinstance(events, list) or not events:
raise ReconstructionValidationError(
f"{location}.development must be a non-empty list when present"
)
for event_index, event in enumerate(events):
event_location = f"{location}.development[{event_index}]"
if not isinstance(event, dict):
raise ReconstructionValidationError(
f"{event_location} must be an object"
)
_expect_exact_keys(
event,
{"event_id", "type", "text", "evidence_refs"},
set(),
event_location,
)
_identifier(event["event_id"], f"{event_location}.event_id", seen)
if event["type"] not in EVENT_TYPES:
raise ReconstructionValidationError(
f"{event_location}.type is invalid: {event['type']!r}"
)
_nonempty_text(event["text"], f"{event_location}.text")
_evidence_refs(
event["evidence_refs"], f"{event_location}.evidence_refs", known
)
if "outcome" in subject:
outcome = subject["outcome"]
outcome_location = f"{location}.outcome"
if not isinstance(outcome, dict):
raise ReconstructionValidationError(
f"{outcome_location} must be a non-empty object when present"
)
_expect_exact_keys(
outcome,
{"text", "scope", "certainty", "evidence_refs"},
set(),
outcome_location,
)
_nonempty_text(outcome["text"], f"{outcome_location}.text")
_nonempty_text(outcome["scope"], f"{outcome_location}.scope")
if outcome["certainty"] not in OUTCOME_CERTAINTIES:
raise ReconstructionValidationError(
f"{outcome_location}.certainty is invalid: {outcome['certainty']!r}"
)
_evidence_refs(
outcome["evidence_refs"], f"{outcome_location}.evidence_refs", known
)
if "actions" in subject:
actions = subject["actions"]
if not isinstance(actions, list) or not actions:
raise ReconstructionValidationError(
f"{location}.actions must be a non-empty list when present"
)
for action_index, action in enumerate(actions):
action_location = f"{location}.actions[{action_index}]"
if not isinstance(action, dict):
raise ReconstructionValidationError(
f"{action_location} must be an object"
)
_expect_exact_keys(
action,
{"action_id", "text", "responsible", "deadline", "evidence_refs"},
set(),
action_location,
)
_identifier(action["action_id"], f"{action_location}.action_id", seen)
_nonempty_text(action["text"], f"{action_location}.text")
for field in ("responsible", "deadline"):
_nullable_text(action[field], f"{action_location}.{field}")
_evidence_refs(
action["evidence_refs"], f"{action_location}.evidence_refs", known
)
if "unresolved_issues" in subject:
issues = subject["unresolved_issues"]
if not isinstance(issues, list) or not issues:
raise ReconstructionValidationError(
f"{location}.unresolved_issues must be a non-empty list when present"
)
for issue_index, issue in enumerate(issues):
issue_location = f"{location}.unresolved_issues[{issue_index}]"
if not isinstance(issue, dict):
raise ReconstructionValidationError(
f"{issue_location} must be an object"
)
_expect_exact_keys(
issue,
{"issue_id", "text", "evidence_refs"},
set(),
issue_location,
)
_identifier(issue["issue_id"], f"{issue_location}.issue_id", seen)
_nonempty_text(issue["text"], f"{issue_location}.text")
_evidence_refs(
issue["evidence_refs"], f"{issue_location}.evidence_refs", known
)
return data
def build_prompt(case: dict[str, Any]) -> str:
evidence_units = case["evidence_units"]
validate_evidence_units(evidence_units)
return PROMPT_TEMPLATE.format(
case_id=case["case_id"],
evidence_json=json.dumps(evidence_units, ensure_ascii=False, indent=2),
)
def parse_model_json(raw_text: str) -> dict[str, Any]:
data = json.loads(raw_text)
if not isinstance(data, dict):
raise ReconstructionValidationError("model response JSON must be an object")
return data
def build_ollama_payload(
model: str, prompt: str, num_ctx: int, num_predict: int
) -> dict[str, Any]:
return {
"model": model,
"prompt": prompt,
"think": False,
"stream": False,
"format": "json",
"options": {
"temperature": 0,
"num_ctx": num_ctx,
"num_predict": num_predict,
},
}
def call_ollama(
endpoint: str,
model: str,
prompt: str,
timeout: int,
num_ctx: int,
num_predict: int,
) -> tuple[str, dict[str, Any]]:
payload = build_ollama_payload(model, prompt, num_ctx, num_predict)
started = time.perf_counter()
response = requests.post(endpoint, json=payload, timeout=timeout)
elapsed = time.perf_counter() - started
response.raise_for_status()
data = response.json()
if not isinstance(data, dict):
raise ValueError("Ollama response must be a JSON object")
raw_text = data.get("response")
if not isinstance(raw_text, str) or not raw_text.strip():
raise ValueError("Ollama returned no usable response text")
metadata = {
"model": data.get("model", model),
"elapsed_seconds": round(elapsed, 3),
"total_duration_ns": data.get("total_duration"),
"load_duration_ns": data.get("load_duration"),
"prompt_eval_count": data.get("prompt_eval_count"),
"prompt_eval_duration_ns": data.get("prompt_eval_duration"),
"eval_count": data.get("eval_count"),
"eval_duration_ns": data.get("eval_duration"),
"configuration": {
"temperature": 0,
"think": False,
"num_ctx": num_ctx,
"num_predict": num_predict,
},
}
return raw_text.strip(), metadata
def _all_text(subjects: list[dict[str, Any]]) -> str:
parts: list[str] = []
for subject in subjects:
parts.append(subject["title"])
for event in subject.get("development", []):
parts.append(event["text"])
outcome = subject.get("outcome")
if outcome:
parts.extend((outcome["text"], outcome["scope"]))
for action in subject.get("actions", []):
parts.append(action["text"])
for issue in subject.get("unresolved_issues", []):
parts.append(issue["text"])
return " ".join(parts).casefold()
def _contains_any(text: str, terms: list[str]) -> bool:
return any(term.casefold() in text for term in terms)
def evaluate_reconstruction(
reconstruction: dict[str, Any], expected: dict[str, Any]
) -> dict[str, Any]:
subjects = reconstruction["subjects"]
combined = _all_text(subjects)
events = [event for subject in subjects for event in subject.get("development", [])]
outcomes = [subject["outcome"] for subject in subjects if "outcome" in subject]
actions = [action for subject in subjects for action in subject.get("actions", [])]
issues = [issue for subject in subjects for issue in subject.get("unresolved_issues", [])]
checks: list[dict[str, Any]] = []
def add(name: str, passed: bool, critical: bool = False) -> None:
checks.append({"name": name, "passed": passed, "critical": critical})
add("subject_count", len(subjects) == expected.get("subject_count", 1))
add("subject_identity", _contains_any(combined, expected["subject_terms"]))
event_types = {event["type"] for event in events}
for event_type in expected.get("required_event_types", []):
add(f"event_type:{event_type}", event_type in event_types)
expected_outcome = expected.get("outcome", {})
outcome_required = expected_outcome.get("required", False)
add(
"outcome_presence",
bool(outcomes) is outcome_required,
critical=not outcome_required and bool(outcomes),
)
if outcome_required and outcomes:
outcome_text = " ".join(
f"{item['text']} {item['scope']}" for item in outcomes
).casefold()
add("outcome_meaning", _contains_any(outcome_text, expected_outcome["terms"]))
add(
"outcome_scope",
_contains_any(outcome_text, expected_outcome.get("scope_terms", [])),
critical=True,
)
add(
"outcome_certainty",
any(
item["certainty"] in expected_outcome.get("certainties", [])
for item in outcomes
),
)
expected_actions = expected.get("actions", {})
minimum_actions = expected_actions.get("minimum", 0)
add(
"action_count",
len(actions) >= minimum_actions if minimum_actions else not actions,
critical=minimum_actions == 0 and bool(actions),
)
if minimum_actions and actions:
action_text = " ".join(item["text"] for item in actions).casefold()
add("action_meaning", _contains_any(action_text, expected_actions["terms"]))
if "responsible" in expected_actions:
add(
"action_responsibility",
any(
item["responsible"] == expected_actions["responsible"]
for item in actions
),
critical=True,
)
expected_issues = expected.get("unresolved", {})
minimum_issues = expected_issues.get("minimum", 0)
add(
"unresolved_count",
len(issues) >= minimum_issues if minimum_issues else not issues,
critical=minimum_issues == 0 and bool(issues),
)
if minimum_issues and issues:
issue_text = " ".join(item["text"] for item in issues).casefold()
add("unresolved_meaning", _contains_any(issue_text, expected_issues["terms"]))
passed = sum(check["passed"] for check in checks)
critical_failures = [
check["name"] for check in checks if check["critical"] and not check["passed"]
]
ratio = passed / len(checks)
if ratio == 1:
verdict = "PASS"
elif ratio >= 0.6 and not critical_failures:
verdict = "PARTIAL"
else:
verdict = "FAIL"
failed = [check["name"] for check in checks if not check["passed"]]
reason = "All semantic checks passed." if not failed else "Failed: " + ", ".join(failed)
return {
"verdict": verdict,
"reason": reason,
"passed_checks": passed,
"check_count": len(checks),
"critical_failures": critical_failures,
"checks": checks,
}
def load_fixture(path: Path) -> list[dict[str, Any]]:
data = json.loads(path.read_text(encoding="utf-8-sig"))
if not isinstance(data, dict) or set(data) != {"cases"}:
raise ValueError("fixture must contain exactly one 'cases' list")
cases = data["cases"]
if not isinstance(cases, list) or not cases:
raise ValueError("fixture cases must be a non-empty list")
seen: set[str] = set()
for index, case in enumerate(cases):
if not isinstance(case, dict):
raise ValueError(f"cases[{index}] must be an object")
required = {"case_id", "description", "evidence_units", "expected"}
if set(case) != required:
raise ValueError(f"cases[{index}] must contain exactly {sorted(required)}")
case_id = _nonempty_text(case["case_id"], f"cases[{index}].case_id")
if case_id in seen:
raise ValueError(f"duplicate case_id: {case_id}")
seen.add(case_id)
_nonempty_text(case["description"], f"cases[{index}].description")
validate_evidence_units(case["evidence_units"])
if not isinstance(case["expected"], dict):
raise ValueError(f"cases[{index}].expected must be an object")
return cases
def run_case(
case: dict[str, Any],
output_root: Path,
endpoint: str,
model: str,
timeout: int,
num_ctx: int,
num_predict: int,
) -> dict[str, Any]:
case_dir = output_root / case["case_id"]
case_dir.mkdir(parents=True, exist_ok=False)
input_payload = {
"case_id": case["case_id"],
"description": case["description"],
"evidence_units": case["evidence_units"],
}
(case_dir / "input.json").write_text(
json.dumps(input_payload, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
prompt = build_prompt(case)
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
started = time.perf_counter()
try:
raw_text, metadata = call_ollama(
endpoint, model, prompt, timeout, num_ctx, num_predict
)
(case_dir / "raw_model_response.txt").write_text(
raw_text + "\n", encoding="utf-8"
)
(case_dir / "ollama_metadata.json").write_text(
json.dumps(metadata, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
parsed = parse_model_json(raw_text)
(case_dir / "parsed_output.json").write_text(
json.dumps(parsed, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
validated = validate_reconstruction(parsed, case["evidence_units"])
evaluation = evaluate_reconstruction(validated, case["expected"])
except requests.RequestException as exc:
failure = {
"case_id": case["case_id"],
"error_type": type(exc).__name__,
"error": str(exc),
"elapsed_seconds": round(time.perf_counter() - started, 3),
}
(case_dir / "validation_failure.json").write_text(
json.dumps(failure, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
raise
except (json.JSONDecodeError, ReconstructionValidationError, ValueError) as exc:
elapsed = round(time.perf_counter() - started, 3)
failure = {
"case_id": case["case_id"],
"error_type": type(exc).__name__,
"error": str(exc),
"elapsed_seconds": elapsed,
}
(case_dir / "validation_failure.json").write_text(
json.dumps(failure, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
result = {
"case_id": case["case_id"],
"description": case["description"],
"verdict": "FAIL",
"reason": f"{type(exc).__name__}: {exc}",
"passed_checks": 0,
"check_count": 0,
"critical_failures": ["schema_validation"],
"checks": [],
"elapsed_seconds": elapsed,
"subject_titles": [],
}
(case_dir / "evaluation.json").write_text(
json.dumps(result, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
return result
result = {
"case_id": case["case_id"],
"description": case["description"],
**evaluation,
"elapsed_seconds": metadata["elapsed_seconds"],
"subject_titles": [item["title"] for item in validated["subjects"]],
}
(case_dir / "evaluation.json").write_text(
json.dumps(result, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
return result
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
cases = load_fixture(args.fixture)
selected = set(args.case_ids or [])
if selected:
known = {case["case_id"] for case in cases}
unknown = selected - known
if unknown:
raise ValueError(f"unknown requested case IDs: {sorted(unknown)}")
cases = [case for case in cases if case["case_id"] in selected]
args.output.mkdir(parents=True, exist_ok=False)
results: list[dict[str, Any]] = []
started = time.perf_counter()
for index, case in enumerate(cases, start=1):
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
results.append(
run_case(
case,
args.output,
args.endpoint,
args.model,
args.timeout,
args.num_ctx,
args.num_predict,
)
)
summary = {
"experiment": "topic_reconstruction_v2",
"schema_version": SCHEMA_VERSION,
"model": args.model,
"temperature": 0,
"think": False,
"case_count": len(cases),
"llm_call_count": len(results),
"runtime_seconds": round(time.perf_counter() - started, 3),
"verdict_counts": {
verdict: sum(item["verdict"] == verdict for item in results)
for verdict in ("PASS", "PARTIAL", "FAIL")
},
"results": results,
}
(args.output / "summary.json").write_text(
json.dumps(summary, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
return summary
def main() -> int:
args = parse_args()
try:
summary = run_experiment(args)
except (OSError, ValueError, requests.RequestException) as exc:
print(f"Error: {exc}")
return 1
print(json.dumps(summary["verdict_counts"], sort_keys=True))
print(f"Artifacts: {args.output.resolve()}")
return 0
if __name__ == "__main__":
raise SystemExit(main())