Add evidence-near semantic architecture experiments
Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
This commit is contained in:
@@ -0,0 +1 @@
|
||||
"""Isolated evidence-near observation experiment."""
|
||||
@@ -0,0 +1,395 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Extract evidence-near observations for a fixed Discussion Subject."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
SCHEMA_VERSION = "experimental-evidence-observations-v1"
|
||||
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
|
||||
DEFAULT_MODEL = "qwen3.5:9B"
|
||||
DEFAULT_TIMEOUT = 300
|
||||
DEFAULT_NUM_CTX = 16384
|
||||
DEFAULT_NUM_PREDICT = 4096
|
||||
|
||||
RELATIONS = {"none", "supports", "opposes", "qualifies", "limits_scope"}
|
||||
MODALITIES = {
|
||||
"factual",
|
||||
"possible",
|
||||
"suggested",
|
||||
"interpersonal_request",
|
||||
"impersonal_necessity",
|
||||
"information_question",
|
||||
"committed",
|
||||
}
|
||||
TEMPORALITIES = {"existing", "future", "completed", "unspecified"}
|
||||
EVALUATIONS = {"positive", "negative", "none"}
|
||||
AGREEMENTS = {"accepted", "rejected", "unclear", "none"}
|
||||
RESPONSIBILITIES = {"none", "named", "accepted"}
|
||||
UNCERTAINTIES = {"present", "absent"}
|
||||
CLARIFICATION_NEEDS = {"explicit", "implicit", "none"}
|
||||
OBSERVATION_ID_RE = re.compile(r"^obs_[1-9][0-9]*$")
|
||||
|
||||
|
||||
class ObservationValidationError(ValueError):
|
||||
"""Raised when an experimental fixture or model output is invalid."""
|
||||
|
||||
|
||||
PROMPT_TEMPLATE = """You extract atomic, evidence-near observations for one fixed Discussion Subject.
|
||||
|
||||
Stop before protocol interpretation. Never classify anything as an idea, proposal,
|
||||
objection, decision, action item, or open question. Do not determine protocol
|
||||
eligibility, reconstruct topics, generate a protocol, or invent missing stages.
|
||||
|
||||
Split an evidence unit into multiple observations when it directly contains multiple
|
||||
propositions. Preserve every observation's source evidence ID. Use concise content in
|
||||
the evidence language.
|
||||
|
||||
Return exactly one JSON object with this shape:
|
||||
{{
|
||||
"schema_version": "experimental-evidence-observations-v1",
|
||||
"subject_id": "copy exactly",
|
||||
"subject": "copy exactly",
|
||||
"observations": [
|
||||
{{
|
||||
"observation_id": "obs_1",
|
||||
"evidence_id": "e1",
|
||||
"content": "directly supported atomic observation",
|
||||
"target": "discussion_subject",
|
||||
"relation": "none",
|
||||
"modality": "factual",
|
||||
"temporality": "existing",
|
||||
"evaluation": "none",
|
||||
"agreement": "none",
|
||||
"responsibility": "none",
|
||||
"person": null,
|
||||
"uncertainty": "absent",
|
||||
"clarification_need": "none",
|
||||
"scope": "absent"
|
||||
}}
|
||||
]
|
||||
}}
|
||||
|
||||
Rules:
|
||||
- Number observation_id sequentially as obs_1, obs_2, ... in evidence order.
|
||||
- target is "discussion_subject", one earlier observation_id, or a non-empty list of
|
||||
earlier observation_ids only when the evidence jointly refers to them.
|
||||
- relation is only none, supports, opposes, qualifies, or limits_scope.
|
||||
- modality is only factual, possible, suggested, interpersonal_request,
|
||||
impersonal_necessity, information_question, or committed.
|
||||
- interpersonal_request is a direct request to another person.
|
||||
- impersonal_necessity says something needs to happen without assigning it.
|
||||
- information_question expresses missing information without assigning work.
|
||||
- temporality is only existing, future, completed, or unspecified.
|
||||
- evaluation is positive, negative, or none. Do not infer evaluation from world
|
||||
knowledge. A bare cost or technical fact normally has evaluation none.
|
||||
- agreement is only accepted, rejected, unclear, or none and applies to target.
|
||||
- responsibility is none, named, or accepted. Use named only for an explicitly
|
||||
addressed candidate and accepted only for explicit acceptance/commitment.
|
||||
- person is the explicit person's name for named/accepted responsibility; otherwise
|
||||
use JSON null. Mentioning or speaking in first person does not establish ownership.
|
||||
- uncertainty is present or absent.
|
||||
- clarification_need is explicit, implicit, or none.
|
||||
- scope is an evidence-grounded qualifier, or exactly "absent". Never use null or the
|
||||
string "null" anywhere.
|
||||
- Confirmation of a rejection targets the rejection observation, not the option.
|
||||
- A trial-only qualification targets and limits the accepted trial.
|
||||
- A negative consequence can oppose another observation without requiring
|
||||
clarification.
|
||||
- Personal preference is not group rejection.
|
||||
- Collective "we" does not name an individual owner.
|
||||
|
||||
Fixed Gold input:
|
||||
{input_json}
|
||||
"""
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description="Run the evidence-observation Gold experiment.")
|
||||
parser.add_argument("fixture", type=Path)
|
||||
parser.add_argument("-o", "--output", type=Path, required=True)
|
||||
parser.add_argument("--model", default=DEFAULT_MODEL)
|
||||
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
|
||||
parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT)
|
||||
parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX)
|
||||
parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None:
|
||||
missing = required - value.keys()
|
||||
unknown = value.keys() - required
|
||||
if missing:
|
||||
raise ObservationValidationError(f"{location} missing required keys: {sorted(missing)}")
|
||||
if unknown:
|
||||
raise ObservationValidationError(f"{location} has unknown keys: {sorted(unknown)}")
|
||||
|
||||
|
||||
def _text(value: Any, location: str) -> str:
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise ObservationValidationError(f"{location} must be a non-empty string")
|
||||
result = value.strip()
|
||||
if result.casefold() == "null":
|
||||
raise ObservationValidationError(f"{location} must not be the string 'null'")
|
||||
return result
|
||||
|
||||
|
||||
OBSERVATION_KEYS = {
|
||||
"observation_id", "evidence_id", "content", "target", "relation", "modality",
|
||||
"temporality", "evaluation", "agreement", "responsibility", "person",
|
||||
"uncertainty", "clarification_need", "scope",
|
||||
}
|
||||
|
||||
|
||||
def _validate_target(value: Any, location: str, earlier: set[str]) -> None:
|
||||
if isinstance(value, str):
|
||||
target = _text(value, location)
|
||||
if target != "discussion_subject" and target not in earlier:
|
||||
raise ObservationValidationError(f"{location} references unknown or later observation: {target}")
|
||||
return
|
||||
if not isinstance(value, list) or not value:
|
||||
raise ObservationValidationError(f"{location} must be discussion_subject, an earlier observation ID, or a non-empty list")
|
||||
if len(value) < 2:
|
||||
raise ObservationValidationError(f"{location} list must contain at least two jointly referenced observations")
|
||||
seen: set[str] = set()
|
||||
for index, item in enumerate(value):
|
||||
target = _text(item, f"{location}[{index}]")
|
||||
if target not in earlier:
|
||||
raise ObservationValidationError(f"{location}[{index}] references unknown or later observation: {target}")
|
||||
if target in seen:
|
||||
raise ObservationValidationError(f"{location} contains duplicate target: {target}")
|
||||
seen.add(target)
|
||||
|
||||
|
||||
def validate_observations(data: Any, case: dict[str, Any]) -> dict[str, Any]:
|
||||
validate_case(case)
|
||||
if not isinstance(data, dict):
|
||||
raise ObservationValidationError("output must be an object")
|
||||
_exact_keys(data, {"schema_version", "subject_id", "subject", "observations"}, "output")
|
||||
if data["schema_version"] != SCHEMA_VERSION:
|
||||
raise ObservationValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
|
||||
if data["subject_id"] != case["subject_id"] or data["subject"] != case["subject"]:
|
||||
raise ObservationValidationError("model changed the fixed Discussion Subject")
|
||||
observations = data["observations"]
|
||||
if not isinstance(observations, list) or not observations:
|
||||
raise ObservationValidationError("output.observations must be a non-empty array")
|
||||
known_evidence = {item["evidence_id"] for item in case["evidence"]}
|
||||
earlier: set[str] = set()
|
||||
for index, observation in enumerate(observations, start=1):
|
||||
location = f"output.observations[{index - 1}]"
|
||||
if not isinstance(observation, dict):
|
||||
raise ObservationValidationError(f"{location} must be an object")
|
||||
_exact_keys(observation, OBSERVATION_KEYS, location)
|
||||
observation_id = _text(observation["observation_id"], f"{location}.observation_id")
|
||||
if not OBSERVATION_ID_RE.fullmatch(observation_id) or observation_id != f"obs_{index}":
|
||||
raise ObservationValidationError(f"{location}.observation_id must be obs_{index}")
|
||||
evidence_id = _text(observation["evidence_id"], f"{location}.evidence_id")
|
||||
if evidence_id not in known_evidence:
|
||||
raise ObservationValidationError(f"{location}.evidence_id references unknown evidence: {evidence_id}")
|
||||
_text(observation["content"], f"{location}.content")
|
||||
_validate_target(observation["target"], f"{location}.target", earlier)
|
||||
for field, values in (
|
||||
("relation", RELATIONS), ("modality", MODALITIES),
|
||||
("temporality", TEMPORALITIES), ("evaluation", EVALUATIONS),
|
||||
("agreement", AGREEMENTS), ("responsibility", RESPONSIBILITIES),
|
||||
("uncertainty", UNCERTAINTIES), ("clarification_need", CLARIFICATION_NEEDS),
|
||||
):
|
||||
if observation[field] not in values:
|
||||
raise ObservationValidationError(f"{location}.{field} is invalid: {observation[field]!r}")
|
||||
person = observation["person"]
|
||||
if observation["responsibility"] == "none":
|
||||
if person is not None:
|
||||
raise ObservationValidationError(f"{location}.person must be JSON null when responsibility is none")
|
||||
else:
|
||||
_text(person, f"{location}.person")
|
||||
scope = _text(observation["scope"], f"{location}.scope")
|
||||
if scope.casefold() == "null":
|
||||
raise ObservationValidationError(f"{location}.scope must use 'absent', not 'null'")
|
||||
earlier.add(observation_id)
|
||||
return data
|
||||
|
||||
|
||||
def validate_case(case: Any) -> dict[str, Any]:
|
||||
if not isinstance(case, dict):
|
||||
raise ObservationValidationError("case must be an object")
|
||||
_exact_keys(case, {"case_id", "description", "subject_id", "subject", "evidence", "expected_observations"}, "case")
|
||||
for field in ("case_id", "description", "subject_id", "subject"):
|
||||
_text(case[field], f"case.{field}")
|
||||
evidence = case["evidence"]
|
||||
if not isinstance(evidence, list) or not evidence:
|
||||
raise ObservationValidationError("case.evidence must be a non-empty array")
|
||||
seen: set[str] = set()
|
||||
for index, unit in enumerate(evidence):
|
||||
location = f"case.evidence[{index}]"
|
||||
if not isinstance(unit, dict):
|
||||
raise ObservationValidationError(f"{location} must be an object")
|
||||
_exact_keys(unit, {"evidence_id", "text"}, location)
|
||||
evidence_id = _text(unit["evidence_id"], f"{location}.evidence_id")
|
||||
if evidence_id in seen:
|
||||
raise ObservationValidationError(f"duplicate evidence ID: {evidence_id}")
|
||||
seen.add(evidence_id)
|
||||
_text(unit["text"], f"{location}.text")
|
||||
expected = case["expected_observations"]
|
||||
if not isinstance(expected, list) or not expected:
|
||||
raise ObservationValidationError("case.expected_observations must be a non-empty array")
|
||||
return case
|
||||
|
||||
|
||||
def validate_fixture_case(case: dict[str, Any]) -> dict[str, Any]:
|
||||
validate_case(case)
|
||||
data = {"schema_version": SCHEMA_VERSION, "subject_id": case["subject_id"], "subject": case["subject"], "observations": case["expected_observations"]}
|
||||
validate_observations(data, case)
|
||||
return case
|
||||
|
||||
|
||||
def build_prompt(case: dict[str, Any]) -> str:
|
||||
validate_fixture_case(case)
|
||||
model_input = {"subject_id": case["subject_id"], "subject": case["subject"], "evidence": case["evidence"]}
|
||||
return PROMPT_TEMPLATE.format(input_json=json.dumps(model_input, ensure_ascii=False, indent=2))
|
||||
|
||||
|
||||
def parse_model_json(raw_text: str) -> dict[str, Any]:
|
||||
data = json.loads(raw_text)
|
||||
if not isinstance(data, dict):
|
||||
raise ObservationValidationError("model response JSON must be an object")
|
||||
return data
|
||||
|
||||
|
||||
def build_ollama_payload(model: str, prompt: str, num_ctx: int, num_predict: int) -> dict[str, Any]:
|
||||
return {"model": model, "prompt": prompt, "think": False, "stream": False, "format": "json", "options": {"temperature": 0, "num_ctx": num_ctx, "num_predict": num_predict}}
|
||||
|
||||
|
||||
def call_ollama(endpoint: str, model: str, prompt: str, timeout: int, num_ctx: int, num_predict: int) -> tuple[str, dict[str, Any]]:
|
||||
started = time.perf_counter()
|
||||
response = requests.post(endpoint, json=build_ollama_payload(model, prompt, num_ctx, num_predict), timeout=timeout)
|
||||
elapsed = time.perf_counter() - started
|
||||
response.raise_for_status()
|
||||
body = response.json()
|
||||
raw_text = body.get("response") if isinstance(body, dict) else None
|
||||
if not isinstance(raw_text, str) or not raw_text.strip():
|
||||
raise ValueError("Ollama returned no usable response text")
|
||||
metadata = {
|
||||
"model": body.get("model", model), "elapsed_seconds": round(elapsed, 3),
|
||||
"total_duration_ns": body.get("total_duration"), "load_duration_ns": body.get("load_duration"),
|
||||
"prompt_eval_count": body.get("prompt_eval_count"), "prompt_eval_duration_ns": body.get("prompt_eval_duration"),
|
||||
"eval_count": body.get("eval_count"), "eval_duration_ns": body.get("eval_duration"),
|
||||
"configuration": {"temperature": 0, "think": False, "num_ctx": num_ctx, "num_predict": num_predict, "retries": 0},
|
||||
}
|
||||
return raw_text.strip(), metadata
|
||||
|
||||
|
||||
COMPARE_FIELDS = ("evidence_id", "target", "relation", "modality", "temporality", "evaluation", "agreement", "responsibility", "person", "uncertainty", "clarification_need")
|
||||
|
||||
|
||||
def _scope_matches(actual: str, expected: str) -> bool:
|
||||
if expected == "absent":
|
||||
return actual == "absent"
|
||||
expected_terms = [term.strip().casefold() for term in expected.split("|")]
|
||||
folded = actual.casefold()
|
||||
return any(term in folded for term in expected_terms)
|
||||
|
||||
|
||||
def evaluate_observations(data: dict[str, Any], expected: list[dict[str, Any]]) -> dict[str, Any]:
|
||||
actual = data["observations"]
|
||||
checks: list[dict[str, Any]] = []
|
||||
pair_count = min(len(actual), len(expected))
|
||||
checks.append({"name": "observation_count", "passed": len(actual) == len(expected), "critical": False})
|
||||
categories = {"missing_observations": max(0, len(expected) - len(actual)), "invented_observations": max(0, len(actual) - len(expected)), "stronger_commitment": 0, "weaker_commitment": 0, "incorrect_targets_relations": 0, "incorrect_responsibility": 0, "incorrect_uncertainty_clarification": 0}
|
||||
commitment_rank = {"factual": 0, "possible": 1, "suggested": 1, "information_question": 1, "impersonal_necessity": 2, "interpersonal_request": 2, "committed": 3}
|
||||
for index in range(pair_count):
|
||||
got, want = actual[index], expected[index]
|
||||
for field in COMPARE_FIELDS:
|
||||
passed = got[field] == want[field]
|
||||
checks.append({"name": f"obs_{index + 1}:{field}", "passed": passed, "critical": field in {"evidence_id", "target", "relation", "modality", "agreement", "responsibility", "person"}})
|
||||
if not passed:
|
||||
if field in {"target", "relation"}: categories["incorrect_targets_relations"] += 1
|
||||
if field in {"responsibility", "person"}: categories["incorrect_responsibility"] += 1
|
||||
if field in {"uncertainty", "clarification_need"}: categories["incorrect_uncertainty_clarification"] += 1
|
||||
scope_ok = _scope_matches(got["scope"], want["scope"])
|
||||
checks.append({"name": f"obs_{index + 1}:scope", "passed": scope_ok, "critical": False})
|
||||
got_rank, want_rank = commitment_rank[got["modality"]], commitment_rank[want["modality"]]
|
||||
if got_rank > want_rank or (want["agreement"] == "none" and got["agreement"] in {"accepted", "rejected"}): categories["stronger_commitment"] += 1
|
||||
if got_rank < want_rank or (want["agreement"] in {"accepted", "rejected"} and got["agreement"] == "none"): categories["weaker_commitment"] += 1
|
||||
passed_count = sum(check["passed"] for check in checks)
|
||||
critical_failures = [check["name"] for check in checks if check["critical"] and not check["passed"]]
|
||||
ratio = passed_count / len(checks)
|
||||
if ratio == 1:
|
||||
verdict = "PASS"
|
||||
elif ratio >= 0.7 and categories["stronger_commitment"] == 0 and categories["incorrect_responsibility"] == 0:
|
||||
verdict = "PARTIAL"
|
||||
else:
|
||||
verdict = "FAIL"
|
||||
return {"verdict": verdict, "matched_checks": passed_count, "check_count": len(checks), "match_ratio": round(ratio, 3), "critical_failures": critical_failures, "error_categories": categories, "checks": checks}
|
||||
|
||||
|
||||
def load_fixture(path: Path) -> list[dict[str, Any]]:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
if not isinstance(data, dict) or set(data) != {"cases"} or not isinstance(data["cases"], list) or not data["cases"]:
|
||||
raise ObservationValidationError("fixture must contain exactly one non-empty cases list")
|
||||
seen: set[str] = set()
|
||||
for case in data["cases"]:
|
||||
validate_fixture_case(case)
|
||||
if case["case_id"] in seen:
|
||||
raise ObservationValidationError(f"duplicate case ID: {case['case_id']}")
|
||||
seen.add(case["case_id"])
|
||||
return data["cases"]
|
||||
|
||||
|
||||
def _write_json(path: Path, value: Any) -> None:
|
||||
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def run_case(case: dict[str, Any], output_root: Path, endpoint: str, model: str, timeout: int, num_ctx: int, num_predict: int) -> dict[str, Any]:
|
||||
case_dir = output_root / case["case_id"]
|
||||
case_dir.mkdir(parents=True, exist_ok=False)
|
||||
_write_json(case_dir / "gold_input.json", {key: case[key] for key in ("case_id", "description", "subject_id", "subject", "evidence")})
|
||||
_write_json(case_dir / "gold_expected_observations.json", case["expected_observations"])
|
||||
prompt = build_prompt(case)
|
||||
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
raw, metadata = call_ollama(endpoint, model, prompt, timeout, num_ctx, num_predict)
|
||||
(case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8")
|
||||
_write_json(case_dir / "ollama_metadata.json", metadata)
|
||||
parsed = parse_model_json(raw)
|
||||
_write_json(case_dir / "parsed_observations.json", parsed)
|
||||
validate_observations(parsed, case)
|
||||
validation = {"valid": True, "error": None}
|
||||
evaluation = evaluate_observations(parsed, case["expected_observations"])
|
||||
except requests.RequestException:
|
||||
raise
|
||||
except (json.JSONDecodeError, ObservationValidationError, ValueError) as exc:
|
||||
validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)}
|
||||
evaluation = {"verdict": "FAIL", "matched_checks": 0, "check_count": 0, "match_ratio": 0, "critical_failures": ["schema_validation"], "error_categories": {}, "checks": []}
|
||||
_write_json(case_dir / "validation_result.json", validation)
|
||||
result = {"case_id": case["case_id"], **evaluation, "elapsed_seconds": round(time.perf_counter() - started, 3)}
|
||||
_write_json(case_dir / "evaluation_result.json", result)
|
||||
return result
|
||||
|
||||
|
||||
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
|
||||
cases = load_fixture(args.fixture)
|
||||
args.output.mkdir(parents=True, exist_ok=False)
|
||||
started = time.perf_counter()
|
||||
results = []
|
||||
for index, case in enumerate(cases, start=1):
|
||||
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
|
||||
results.append(run_case(case, args.output, args.endpoint, args.model, args.timeout, args.num_ctx, args.num_predict))
|
||||
summary = {"experiment": "evidence_near_observation_extraction", "schema_version": SCHEMA_VERSION, "model": args.model, "temperature": 0, "think": False, "retries": 0, "case_count": len(cases), "llm_call_count": len(results), "runtime_seconds": round(time.perf_counter() - started, 3), "verdict_counts": {v: sum(r["verdict"] == v for r in results) for v in ("PASS", "PARTIAL", "FAIL")}, "results": results}
|
||||
_write_json(args.output / "summary.json", summary)
|
||||
return summary
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
summary = run_experiment(args)
|
||||
print(json.dumps(summary, ensure_ascii=False, indent=2))
|
||||
return 0 if summary["verdict_counts"]["FAIL"] == 0 else 1
|
||||
@@ -0,0 +1 @@
|
||||
"""Reduced-semantic-load evidence observation experiment."""
|
||||
@@ -0,0 +1,344 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Extract reduced-semantic-load evidence-near observations."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
SCHEMA_VERSION = "experimental-evidence-observations-v2"
|
||||
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
|
||||
DEFAULT_MODEL = "qwen3.5:9B"
|
||||
MODALITIES = {"factual", "possible", "suggested", "interpersonal_request", "impersonal_necessity", "information_question", "committed"}
|
||||
TEMPORALITIES = {"existing", "future", "completed", "unspecified"}
|
||||
EVALUATIONS = {"positive", "negative", "none"}
|
||||
BINARY_SIGNALS = {"explicit", "absent"}
|
||||
PRESENCE_SIGNALS = {"present", "absent"}
|
||||
CLARIFICATION_NEEDS = {"explicit", "implicit", "none"}
|
||||
OBSERVATION_ID_RE = re.compile(r"^obs_[1-9][0-9]*$")
|
||||
|
||||
|
||||
class ObservationValidationError(ValueError):
|
||||
"""Raised for invalid fixtures or model output."""
|
||||
|
||||
|
||||
PROMPT_TEMPLATE = """You extract atomic linguistic and discourse observations for one fixed Discussion Subject.
|
||||
|
||||
Preserve only facts directly expressed by the evidence. Do not derive responsibility,
|
||||
agreement, decisions, action items, open questions, accepted trials, rejected
|
||||
alternatives, established actions, or protocol eligibility. Speaker identity, a name,
|
||||
an addressee, first-person language, collective "we", and impersonal "man" never by
|
||||
themselves establish responsibility.
|
||||
|
||||
Return exactly one JSON object with this shape:
|
||||
{{
|
||||
"schema_version": "experimental-evidence-observations-v2",
|
||||
"subject_id": "copy exactly",
|
||||
"subject": "copy exactly",
|
||||
"observations": [
|
||||
{{
|
||||
"observation_id": "obs_1",
|
||||
"evidence_id": "e1",
|
||||
"content": "directly supported atomic observation",
|
||||
"refers_to": null,
|
||||
"speaker": "name copied from evidence or null",
|
||||
"named_person": null,
|
||||
"addressee": null,
|
||||
"self_reference": false,
|
||||
"collective_we": false,
|
||||
"impersonal_person_reference": false,
|
||||
"modality": "factual",
|
||||
"temporality": "existing",
|
||||
"evaluation": "none",
|
||||
"affirmation": "absent",
|
||||
"negation": "absent",
|
||||
"determination_statement": "absent",
|
||||
"uncertainty": "absent",
|
||||
"clarification_need": "none",
|
||||
"qualifier": null,
|
||||
"limits_target": null
|
||||
}}
|
||||
]
|
||||
}}
|
||||
|
||||
Rules:
|
||||
- Produce multiple observations for distinct propositions in one evidence unit, but do
|
||||
not fragment a single proposition unnecessarily.
|
||||
- observation_id is sequential in evidence order. evidence_id must be copied exactly.
|
||||
- refers_to is null or one earlier observation_id when the utterance explicitly refers
|
||||
to it. Never use arrays. Preserve joint-reference utterances without inventing a
|
||||
multi-target graph.
|
||||
- speaker is the explicit transcript speaker. named_person is a person explicitly
|
||||
named in the proposition. addressee is a person explicitly addressed.
|
||||
- self_reference marks singular first-person self-reference. collective_we marks
|
||||
collective first-person language. impersonal_person_reference marks impersonal
|
||||
person expressions such as German "man".
|
||||
- modality is factual, possible, suggested, interpersonal_request,
|
||||
impersonal_necessity, information_question, or committed.
|
||||
- temporality is existing, future, completed, or unspecified.
|
||||
- evaluation is positive, negative, or none, only when linguistically supported.
|
||||
- affirmation is explicit only for an explicit affirmative discourse signal such as
|
||||
"ja". negation is explicit only for directly expressed negation/rejection.
|
||||
- determination_statement is present only when the utterance explicitly says a
|
||||
determination has been made.
|
||||
- uncertainty is present or absent. clarification_need is explicit, implicit, or none.
|
||||
- qualifier is null or concise evidence-grounded qualifying text.
|
||||
- limits_target is null or one earlier observation explicitly limited in validity or
|
||||
scope by this observation.
|
||||
- Use JSON null, never the string "null". Output no fields beyond the schema.
|
||||
|
||||
Fixed Gold input:
|
||||
{input_json}
|
||||
"""
|
||||
|
||||
|
||||
OBSERVATION_KEYS = {
|
||||
"observation_id", "evidence_id", "content", "refers_to", "speaker",
|
||||
"named_person", "addressee", "self_reference", "collective_we",
|
||||
"impersonal_person_reference", "modality", "temporality", "evaluation",
|
||||
"affirmation", "negation", "determination_statement", "uncertainty",
|
||||
"clarification_need", "qualifier", "limits_target",
|
||||
}
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("fixture", type=Path)
|
||||
parser.add_argument("-o", "--output", type=Path, required=True)
|
||||
parser.add_argument("--model", default=DEFAULT_MODEL)
|
||||
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
|
||||
parser.add_argument("--timeout", type=int, default=300)
|
||||
parser.add_argument("--num-ctx", type=int, default=16384)
|
||||
parser.add_argument("--num-predict", type=int, default=4096)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None:
|
||||
missing, unknown = required - value.keys(), value.keys() - required
|
||||
if missing:
|
||||
raise ObservationValidationError(f"{location} missing required keys: {sorted(missing)}")
|
||||
if unknown:
|
||||
raise ObservationValidationError(f"{location} has unknown keys: {sorted(unknown)}")
|
||||
|
||||
|
||||
def _text(value: Any, location: str) -> str:
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise ObservationValidationError(f"{location} must be a non-empty string")
|
||||
result = value.strip()
|
||||
if result.casefold() == "null":
|
||||
raise ObservationValidationError(f"{location} must not be the string 'null'")
|
||||
return result
|
||||
|
||||
|
||||
def _nullable_text(value: Any, location: str) -> None:
|
||||
if value is not None:
|
||||
_text(value, location)
|
||||
|
||||
|
||||
def _prior_reference(value: Any, location: str, earlier: set[str]) -> None:
|
||||
if value is None:
|
||||
return
|
||||
reference = _text(value, location)
|
||||
if reference not in earlier:
|
||||
raise ObservationValidationError(f"{location} references unknown or later observation: {reference}")
|
||||
|
||||
|
||||
def validate_observations(data: Any, case: dict[str, Any]) -> dict[str, Any]:
|
||||
validate_case(case)
|
||||
if not isinstance(data, dict):
|
||||
raise ObservationValidationError("output must be an object")
|
||||
_exact_keys(data, {"schema_version", "subject_id", "subject", "observations"}, "output")
|
||||
if data["schema_version"] != SCHEMA_VERSION:
|
||||
raise ObservationValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
|
||||
if data["subject_id"] != case["subject_id"] or data["subject"] != case["subject"]:
|
||||
raise ObservationValidationError("model changed the fixed Discussion Subject")
|
||||
observations = data["observations"]
|
||||
if not isinstance(observations, list) or not observations:
|
||||
raise ObservationValidationError("output.observations must be a non-empty array")
|
||||
known_evidence = {item["evidence_id"] for item in case["evidence"]}
|
||||
earlier: set[str] = set()
|
||||
for index, observation in enumerate(observations, 1):
|
||||
location = f"output.observations[{index - 1}]"
|
||||
if not isinstance(observation, dict):
|
||||
raise ObservationValidationError(f"{location} must be an object")
|
||||
_exact_keys(observation, OBSERVATION_KEYS, location)
|
||||
observation_id = _text(observation["observation_id"], f"{location}.observation_id")
|
||||
if not OBSERVATION_ID_RE.fullmatch(observation_id) or observation_id != f"obs_{index}":
|
||||
raise ObservationValidationError(f"{location}.observation_id must be obs_{index}")
|
||||
evidence_id = _text(observation["evidence_id"], f"{location}.evidence_id")
|
||||
if evidence_id not in known_evidence:
|
||||
raise ObservationValidationError(f"{location}.evidence_id references unknown evidence: {evidence_id}")
|
||||
_text(observation["content"], f"{location}.content")
|
||||
_prior_reference(observation["refers_to"], f"{location}.refers_to", earlier)
|
||||
_prior_reference(observation["limits_target"], f"{location}.limits_target", earlier)
|
||||
for field in ("speaker", "named_person", "addressee", "qualifier"):
|
||||
_nullable_text(observation[field], f"{location}.{field}")
|
||||
for field in ("self_reference", "collective_we", "impersonal_person_reference"):
|
||||
if not isinstance(observation[field], bool):
|
||||
raise ObservationValidationError(f"{location}.{field} must be boolean")
|
||||
for field, values in (
|
||||
("modality", MODALITIES), ("temporality", TEMPORALITIES),
|
||||
("evaluation", EVALUATIONS), ("affirmation", BINARY_SIGNALS),
|
||||
("negation", BINARY_SIGNALS), ("determination_statement", PRESENCE_SIGNALS),
|
||||
("uncertainty", PRESENCE_SIGNALS), ("clarification_need", CLARIFICATION_NEEDS),
|
||||
):
|
||||
if observation[field] not in values:
|
||||
raise ObservationValidationError(f"{location}.{field} is invalid: {observation[field]!r}")
|
||||
earlier.add(observation_id)
|
||||
return data
|
||||
|
||||
|
||||
def validate_case(case: Any) -> dict[str, Any]:
|
||||
if not isinstance(case, dict):
|
||||
raise ObservationValidationError("case must be an object")
|
||||
_exact_keys(case, {"case_id", "description", "subject_id", "subject", "evidence", "expected_observations"}, "case")
|
||||
for field in ("case_id", "description", "subject_id", "subject"):
|
||||
_text(case[field], f"case.{field}")
|
||||
if not isinstance(case["evidence"], list) or not case["evidence"]:
|
||||
raise ObservationValidationError("case.evidence must be a non-empty array")
|
||||
seen: set[str] = set()
|
||||
for index, unit in enumerate(case["evidence"]):
|
||||
_exact_keys(unit, {"evidence_id", "text"}, f"case.evidence[{index}]")
|
||||
evidence_id = _text(unit["evidence_id"], f"case.evidence[{index}].evidence_id")
|
||||
if evidence_id in seen:
|
||||
raise ObservationValidationError(f"duplicate evidence ID: {evidence_id}")
|
||||
seen.add(evidence_id)
|
||||
_text(unit["text"], f"case.evidence[{index}].text")
|
||||
if not isinstance(case["expected_observations"], list) or not case["expected_observations"]:
|
||||
raise ObservationValidationError("case.expected_observations must be a non-empty array")
|
||||
return case
|
||||
|
||||
|
||||
def validate_fixture_case(case: dict[str, Any]) -> dict[str, Any]:
|
||||
validate_case(case)
|
||||
validate_observations({"schema_version": SCHEMA_VERSION, "subject_id": case["subject_id"], "subject": case["subject"], "observations": case["expected_observations"]}, case)
|
||||
return case
|
||||
|
||||
|
||||
def build_prompt(case: dict[str, Any]) -> str:
|
||||
validate_fixture_case(case)
|
||||
model_input = {key: case[key] for key in ("subject_id", "subject", "evidence")}
|
||||
return PROMPT_TEMPLATE.format(input_json=json.dumps(model_input, ensure_ascii=False, indent=2))
|
||||
|
||||
|
||||
def parse_model_json(raw_text: str) -> dict[str, Any]:
|
||||
data = json.loads(raw_text)
|
||||
if not isinstance(data, dict):
|
||||
raise ObservationValidationError("model response JSON must be an object")
|
||||
return data
|
||||
|
||||
|
||||
def build_ollama_payload(model: str, prompt: str, num_ctx: int, num_predict: int) -> dict[str, Any]:
|
||||
return {"model": model, "prompt": prompt, "think": False, "stream": False, "format": "json", "options": {"temperature": 0, "num_ctx": num_ctx, "num_predict": num_predict}}
|
||||
|
||||
|
||||
def call_ollama(endpoint: str, model: str, prompt: str, timeout: int, num_ctx: int, num_predict: int) -> tuple[str, dict[str, Any]]:
|
||||
started = time.perf_counter()
|
||||
response = requests.post(endpoint, json=build_ollama_payload(model, prompt, num_ctx, num_predict), timeout=timeout)
|
||||
elapsed = time.perf_counter() - started
|
||||
response.raise_for_status()
|
||||
body = response.json()
|
||||
raw = body.get("response") if isinstance(body, dict) else None
|
||||
if not isinstance(raw, str) or not raw.strip():
|
||||
raise ValueError("Ollama returned no usable response text")
|
||||
metadata = {"model": body.get("model", model), "elapsed_seconds": round(elapsed, 3), "total_duration_ns": body.get("total_duration"), "load_duration_ns": body.get("load_duration"), "prompt_eval_count": body.get("prompt_eval_count"), "prompt_eval_duration_ns": body.get("prompt_eval_duration"), "eval_count": body.get("eval_count"), "eval_duration_ns": body.get("eval_duration"), "configuration": {"temperature": 0, "think": False, "num_ctx": num_ctx, "num_predict": num_predict, "retries": 0}}
|
||||
return raw.strip(), metadata
|
||||
|
||||
|
||||
COMPARE_FIELDS = tuple(sorted(OBSERVATION_KEYS - {"observation_id", "content", "qualifier"}))
|
||||
|
||||
|
||||
def _qualifier_matches(actual: str | None, expected: str | None) -> bool:
|
||||
if expected is None:
|
||||
return actual is None
|
||||
if actual is None:
|
||||
return False
|
||||
return any(term.strip().casefold() in actual.casefold() for term in expected.split("|"))
|
||||
|
||||
|
||||
def evaluate_observations(data: dict[str, Any], expected: list[dict[str, Any]]) -> dict[str, Any]:
|
||||
actual = data["observations"]
|
||||
checks = [{"name": "observation_count", "passed": len(actual) == len(expected), "critical": False}]
|
||||
for index, (got, want) in enumerate(zip(actual, expected), 1):
|
||||
for field in COMPARE_FIELDS:
|
||||
checks.append({"name": f"obs_{index}:{field}", "passed": got[field] == want[field], "critical": field in {"evidence_id", "refers_to", "limits_target", "modality", "affirmation", "negation", "determination_statement"}})
|
||||
checks.append({"name": f"obs_{index}:qualifier", "passed": _qualifier_matches(got["qualifier"], want["qualifier"]), "critical": False})
|
||||
passed = sum(check["passed"] for check in checks)
|
||||
ratio = passed / len(checks)
|
||||
critical = [check["name"] for check in checks if check["critical"] and not check["passed"]]
|
||||
verdict = "PASS" if ratio == 1 else "PARTIAL" if ratio >= 0.75 and not critical else "FAIL"
|
||||
return {"verdict": verdict, "matched_checks": passed, "check_count": len(checks), "match_ratio": round(ratio, 3), "critical_failures": critical, "checks": checks}
|
||||
|
||||
|
||||
def load_fixture(path: Path) -> list[dict[str, Any]]:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
if not isinstance(data, dict) or set(data) != {"cases"} or not isinstance(data["cases"], list) or not data["cases"]:
|
||||
raise ObservationValidationError("fixture must contain exactly one non-empty cases list")
|
||||
seen: set[str] = set()
|
||||
for case in data["cases"]:
|
||||
validate_fixture_case(case)
|
||||
if case["case_id"] in seen:
|
||||
raise ObservationValidationError(f"duplicate case ID: {case['case_id']}")
|
||||
seen.add(case["case_id"])
|
||||
return data["cases"]
|
||||
|
||||
|
||||
def _write_json(path: Path, value: Any) -> None:
|
||||
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def run_case(case: dict[str, Any], output_root: Path, endpoint: str, model: str, timeout: int, num_ctx: int, num_predict: int) -> dict[str, Any]:
|
||||
case_dir = output_root / case["case_id"]
|
||||
case_dir.mkdir(parents=True, exist_ok=False)
|
||||
_write_json(case_dir / "gold_input.json", {key: case[key] for key in ("case_id", "description", "subject_id", "subject", "evidence")})
|
||||
_write_json(case_dir / "gold_expected_observations.json", case["expected_observations"])
|
||||
prompt = build_prompt(case)
|
||||
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
|
||||
started = time.perf_counter()
|
||||
raw, metadata = call_ollama(endpoint, model, prompt, timeout, num_ctx, num_predict)
|
||||
(case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8")
|
||||
_write_json(case_dir / "ollama_metadata.json", metadata)
|
||||
try:
|
||||
parsed = parse_model_json(raw)
|
||||
_write_json(case_dir / "parsed_observations.json", parsed)
|
||||
validate_observations(parsed, case)
|
||||
validation = {"valid": True, "error": None}
|
||||
evaluation = evaluate_observations(parsed, case["expected_observations"])
|
||||
except (json.JSONDecodeError, ObservationValidationError, ValueError) as exc:
|
||||
validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)}
|
||||
evaluation = {"verdict": "FAIL", "matched_checks": 0, "check_count": 0, "match_ratio": 0, "critical_failures": ["schema_validation"], "checks": []}
|
||||
_write_json(case_dir / "validation_result.json", validation)
|
||||
result = {"case_id": case["case_id"], **evaluation, "elapsed_seconds": round(time.perf_counter() - started, 3)}
|
||||
_write_json(case_dir / "evaluation_result.json", result)
|
||||
return result
|
||||
|
||||
|
||||
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
|
||||
cases = load_fixture(args.fixture)
|
||||
args.output.mkdir(parents=True, exist_ok=False)
|
||||
started = time.perf_counter()
|
||||
results = []
|
||||
for index, case in enumerate(cases, 1):
|
||||
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
|
||||
results.append(run_case(case, args.output, args.endpoint, args.model, args.timeout, args.num_ctx, args.num_predict))
|
||||
summary = {"experiment": "evidence_near_observation_extraction_v2", "schema_version": SCHEMA_VERSION, "model": args.model, "temperature": 0, "think": False, "retries": 0, "case_count": len(cases), "llm_call_count": len(results), "runtime_seconds": round(time.perf_counter() - started, 3), "verdict_counts": {verdict: sum(result["verdict"] == verdict for result in results) for verdict in ("PASS", "PARTIAL", "FAIL")}, "results": results}
|
||||
_write_json(args.output / "summary.json", summary)
|
||||
return summary
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
summary = run_experiment(args)
|
||||
print(json.dumps(summary, ensure_ascii=False, indent=2))
|
||||
return 0 if summary["verdict_counts"]["FAIL"] == 0 else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1 @@
|
||||
"""Minimal semantic-preservation observation experiment."""
|
||||
@@ -0,0 +1,271 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Preserve meeting meaning as minimal atomic natural-language observations."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
SCHEMA_VERSION = "experimental-evidence-observations-v3"
|
||||
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
|
||||
DEFAULT_MODEL = "qwen3.5:9B"
|
||||
OBSERVATION_ID_RE = re.compile(r"^obs_[1-9][0-9]*$")
|
||||
OBSERVATION_KEYS = {"observation_id", "evidence_id", "content", "speaker", "named_person", "addressee"}
|
||||
|
||||
|
||||
class ObservationValidationError(ValueError):
|
||||
"""Raised for invalid fixtures or model output."""
|
||||
|
||||
|
||||
PROMPT_TEMPLATE = """Preserve the meeting meaning in atomic natural-language observations.
|
||||
|
||||
This is semantic preservation, not classification or summarization. Return only facts
|
||||
faithfully contributed by the evidence. Conservative wording is more important than
|
||||
elegant prose. When in doubt, preserve the source wording closely.
|
||||
|
||||
Return exactly one JSON object:
|
||||
{{
|
||||
"schema_version": "experimental-evidence-observations-v3",
|
||||
"subject_id": "copy exactly",
|
||||
"subject": "copy exactly",
|
||||
"observations": [
|
||||
{{
|
||||
"observation_id": "obs_1",
|
||||
"evidence_id": "e1",
|
||||
"content": "atomic, semantically faithful observation",
|
||||
"speaker": "speaker copied from evidence",
|
||||
"named_person": null,
|
||||
"addressee": null
|
||||
}}
|
||||
]
|
||||
}}
|
||||
|
||||
Rules:
|
||||
- Use only the six observation fields shown. Do not output classifications, labels,
|
||||
relations, scope fields, responsibility, agreement, decisions, actions, questions,
|
||||
eligibility, or any other field.
|
||||
- observation_id is sequential in evidence order. Copy evidence_id and speaker.
|
||||
- named_person is null or a person explicitly named in that observation's evidence.
|
||||
- addressee is null or a person explicitly addressed in that observation's evidence.
|
||||
- A name, speaker, or addressee never implies responsibility, acceptance, ownership,
|
||||
or assignment.
|
||||
- content is not a summary. Preserve distinctions needed for later interpretation:
|
||||
maybe/perhaps; can/could; should/must; personal, collective, or impersonal wording;
|
||||
explicit requests, acceptances, and rejections; uncertainty and unresolved status;
|
||||
conditions such as "if at all"; quantities; deadlines; trial/process/comparison
|
||||
boundaries; "not yet"; and sequence such as "then".
|
||||
- Never strengthen modality, weaken uncertainty, turn possibility into fact, turn a
|
||||
preference into group rejection, turn a request into established work, turn "we"
|
||||
into individual ownership, remove conditions/limits, generalize, or invent relations.
|
||||
- Split one evidence unit only when it contributes propositions that may later require
|
||||
different interpretations. Do not split merely because it has several clauses.
|
||||
- Do not emit observation-ID relations. When evidence clearly makes an observation
|
||||
depend on the immediately preceding proposition, state that dependency naturally in
|
||||
content, without inventing an antecedent.
|
||||
- Preserve content in the evidence language. Use JSON null, never the string "null".
|
||||
|
||||
Fixed input:
|
||||
{input_json}
|
||||
"""
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("fixture", type=Path)
|
||||
parser.add_argument("-o", "--output", type=Path, required=True)
|
||||
parser.add_argument("--model", default=DEFAULT_MODEL)
|
||||
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
|
||||
parser.add_argument("--timeout", type=int, default=300)
|
||||
parser.add_argument("--num-ctx", type=int, default=16384)
|
||||
parser.add_argument("--num-predict", type=int, default=4096)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None:
|
||||
missing, unknown = required - value.keys(), value.keys() - required
|
||||
if missing:
|
||||
raise ObservationValidationError(f"{location} missing required keys: {sorted(missing)}")
|
||||
if unknown:
|
||||
raise ObservationValidationError(f"{location} has unknown keys: {sorted(unknown)}")
|
||||
|
||||
|
||||
def _text(value: Any, location: str) -> str:
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise ObservationValidationError(f"{location} must be a non-empty string")
|
||||
result = value.strip()
|
||||
if result.casefold() == "null":
|
||||
raise ObservationValidationError(f"{location} must not be the string 'null'")
|
||||
return result
|
||||
|
||||
|
||||
def _explicit_people(text: str) -> set[str]:
|
||||
prefix = text.split(":", 1)[0].strip() if ":" in text else ""
|
||||
candidates = set(re.findall(r"\b(?:Dr\.\s+)?[A-ZÄÖÜ][A-Za-zÄÖÜäöüß-]+(?:\s+[A-ZÄÖÜ][A-Za-zÄÖÜäöüß-]+)*", text))
|
||||
candidates.discard(prefix)
|
||||
return candidates
|
||||
|
||||
|
||||
def validate_observations(data: Any, case: dict[str, Any]) -> dict[str, Any]:
|
||||
validate_case(case)
|
||||
if not isinstance(data, dict):
|
||||
raise ObservationValidationError("output must be an object")
|
||||
_exact_keys(data, {"schema_version", "subject_id", "subject", "observations"}, "output")
|
||||
if data["schema_version"] != SCHEMA_VERSION:
|
||||
raise ObservationValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
|
||||
if data["subject_id"] != case["subject_id"] or data["subject"] != case["subject"]:
|
||||
raise ObservationValidationError("model changed the fixed Discussion Subject")
|
||||
observations = data["observations"]
|
||||
if not isinstance(observations, list) or not observations:
|
||||
raise ObservationValidationError("output.observations must be a non-empty list")
|
||||
evidence = {item["evidence_id"]: item["text"] for item in case["evidence"]}
|
||||
seen: set[str] = set()
|
||||
for index, observation in enumerate(observations):
|
||||
location = f"output.observations[{index}]"
|
||||
if not isinstance(observation, dict):
|
||||
raise ObservationValidationError(f"{location} must be an object")
|
||||
_exact_keys(observation, OBSERVATION_KEYS, location)
|
||||
observation_id = _text(observation["observation_id"], f"{location}.observation_id")
|
||||
if not OBSERVATION_ID_RE.fullmatch(observation_id) or observation_id in seen:
|
||||
raise ObservationValidationError(f"{location}.observation_id must be unique and match obs_N")
|
||||
seen.add(observation_id)
|
||||
evidence_id = _text(observation["evidence_id"], f"{location}.evidence_id")
|
||||
if evidence_id not in evidence:
|
||||
raise ObservationValidationError(f"{location}.evidence_id references unknown evidence: {evidence_id}")
|
||||
source = evidence[evidence_id]
|
||||
source_speaker = source.split(":", 1)[0].strip()
|
||||
speaker = _text(observation["speaker"], f"{location}.speaker")
|
||||
if speaker != source_speaker:
|
||||
raise ObservationValidationError(f"{location}.speaker must match evidence speaker {source_speaker!r}")
|
||||
_text(observation["content"], f"{location}.content")
|
||||
explicit_people = _explicit_people(source)
|
||||
for field in ("named_person", "addressee"):
|
||||
person = observation[field]
|
||||
if person is not None:
|
||||
person = _text(person, f"{location}.{field}")
|
||||
if person not in explicit_people:
|
||||
raise ObservationValidationError(f"{location}.{field} is not an explicit person in evidence: {person!r}")
|
||||
return data
|
||||
|
||||
|
||||
def validate_case(case: Any) -> dict[str, Any]:
|
||||
required = {"case_id", "description", "subject_id", "subject", "evidence", "semantic_requirements"}
|
||||
if not isinstance(case, dict):
|
||||
raise ObservationValidationError("case must be an object")
|
||||
_exact_keys(case, required, "case")
|
||||
for field in ("case_id", "description", "subject_id", "subject"):
|
||||
_text(case[field], f"case.{field}")
|
||||
if not isinstance(case["evidence"], list) or not case["evidence"]:
|
||||
raise ObservationValidationError("case.evidence must be a non-empty list")
|
||||
evidence_ids: set[str] = set()
|
||||
for index, unit in enumerate(case["evidence"]):
|
||||
_exact_keys(unit, {"evidence_id", "text"}, f"case.evidence[{index}]")
|
||||
evidence_id = _text(unit["evidence_id"], f"case.evidence[{index}].evidence_id")
|
||||
if evidence_id in evidence_ids:
|
||||
raise ObservationValidationError(f"duplicate evidence ID: {evidence_id}")
|
||||
evidence_ids.add(evidence_id)
|
||||
_text(unit["text"], f"case.evidence[{index}].text")
|
||||
if not isinstance(case["semantic_requirements"], list) or not case["semantic_requirements"]:
|
||||
raise ObservationValidationError("case.semantic_requirements must be a non-empty list")
|
||||
for index, requirement in enumerate(case["semantic_requirements"]):
|
||||
_text(requirement, f"case.semantic_requirements[{index}]")
|
||||
return case
|
||||
|
||||
|
||||
def build_prompt(case: dict[str, Any]) -> str:
|
||||
validate_case(case)
|
||||
model_input = {key: case[key] for key in ("subject_id", "subject", "evidence")}
|
||||
return PROMPT_TEMPLATE.format(input_json=json.dumps(model_input, ensure_ascii=False, indent=2))
|
||||
|
||||
|
||||
def parse_model_json(raw_text: str) -> dict[str, Any]:
|
||||
data = json.loads(raw_text)
|
||||
if not isinstance(data, dict):
|
||||
raise ObservationValidationError("model response JSON must be an object")
|
||||
return data
|
||||
|
||||
|
||||
def build_ollama_payload(model: str, prompt: str, num_ctx: int, num_predict: int) -> dict[str, Any]:
|
||||
return {"model": model, "prompt": prompt, "think": False, "stream": False, "format": "json", "options": {"temperature": 0, "num_ctx": num_ctx, "num_predict": num_predict}}
|
||||
|
||||
|
||||
def call_ollama(endpoint: str, model: str, prompt: str, timeout: int, num_ctx: int, num_predict: int) -> tuple[str, dict[str, Any]]:
|
||||
started = time.perf_counter()
|
||||
response = requests.post(endpoint, json=build_ollama_payload(model, prompt, num_ctx, num_predict), timeout=timeout)
|
||||
elapsed = time.perf_counter() - started
|
||||
response.raise_for_status()
|
||||
body = response.json()
|
||||
raw = body.get("response") if isinstance(body, dict) else None
|
||||
if not isinstance(raw, str) or not raw.strip():
|
||||
raise ValueError("Ollama returned no usable response text")
|
||||
metadata = {"model": body.get("model", model), "elapsed_seconds": round(elapsed, 3), "total_duration_ns": body.get("total_duration"), "load_duration_ns": body.get("load_duration"), "prompt_eval_count": body.get("prompt_eval_count"), "prompt_eval_duration_ns": body.get("prompt_eval_duration"), "eval_count": body.get("eval_count"), "eval_duration_ns": body.get("eval_duration"), "configuration": {"temperature": 0, "think": False, "num_ctx": num_ctx, "num_predict": num_predict, "retries": 0}}
|
||||
return raw.strip(), metadata
|
||||
|
||||
|
||||
def load_fixture(path: Path) -> list[dict[str, Any]]:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
if not isinstance(data, dict) or set(data) != {"cases"} or not isinstance(data["cases"], list) or not data["cases"]:
|
||||
raise ObservationValidationError("fixture must contain exactly one non-empty cases list")
|
||||
seen: set[str] = set()
|
||||
for case in data["cases"]:
|
||||
validate_case(case)
|
||||
if case["case_id"] in seen:
|
||||
raise ObservationValidationError(f"duplicate case ID: {case['case_id']}")
|
||||
seen.add(case["case_id"])
|
||||
return data["cases"]
|
||||
|
||||
|
||||
def _write_json(path: Path, value: Any) -> None:
|
||||
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def run_case(case: dict[str, Any], output_root: Path, endpoint: str, model: str, timeout: int, num_ctx: int, num_predict: int) -> dict[str, Any]:
|
||||
case_dir = output_root / case["case_id"]
|
||||
case_dir.mkdir(parents=True, exist_ok=False)
|
||||
_write_json(case_dir / "source_evidence.json", {key: case[key] for key in ("case_id", "description", "subject_id", "subject", "evidence")})
|
||||
_write_json(case_dir / "gold_semantic_requirements.json", case["semantic_requirements"])
|
||||
prompt = build_prompt(case)
|
||||
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
|
||||
started = time.perf_counter()
|
||||
raw, metadata = call_ollama(endpoint, model, prompt, timeout, num_ctx, num_predict)
|
||||
(case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8")
|
||||
_write_json(case_dir / "ollama_metadata.json", metadata)
|
||||
try:
|
||||
parsed = parse_model_json(raw)
|
||||
_write_json(case_dir / "parsed_observations.json", parsed)
|
||||
validate_observations(parsed, case)
|
||||
validation = {"valid": True, "error": None}
|
||||
except (json.JSONDecodeError, ObservationValidationError, ValueError) as exc:
|
||||
validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)}
|
||||
_write_json(case_dir / "structural_validation.json", validation)
|
||||
return {"case_id": case["case_id"], "structurally_valid": validation["valid"], "elapsed_seconds": round(time.perf_counter() - started, 3)}
|
||||
|
||||
|
||||
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
|
||||
cases = load_fixture(args.fixture)
|
||||
args.output.mkdir(parents=True, exist_ok=False)
|
||||
started = time.perf_counter()
|
||||
results = []
|
||||
for index, case in enumerate(cases, 1):
|
||||
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
|
||||
results.append(run_case(case, args.output, args.endpoint, args.model, args.timeout, args.num_ctx, args.num_predict))
|
||||
summary = {"experiment": "evidence_near_observation_extraction_v3", "schema_version": SCHEMA_VERSION, "model": args.model, "temperature": 0, "think": False, "retries": 0, "case_count": len(cases), "llm_call_count": len(results), "runtime_seconds": round(time.perf_counter() - started, 3), "structurally_valid_count": sum(result["structurally_valid"] for result in results), "results": results}
|
||||
_write_json(args.output / "summary.json", summary)
|
||||
return summary
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
summary = run_experiment(args)
|
||||
print(json.dumps(summary, ensure_ascii=False, indent=2))
|
||||
return 0 if summary["structurally_valid_count"] == summary["case_count"] else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1 @@
|
||||
"""Isolated experimental semantic synthesis for known discussion subjects."""
|
||||
@@ -0,0 +1,640 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run semantic synthesis with subject detection and evidence assignment fixed."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
SCHEMA_VERSION = "experimental-semantic-synthesis-v1"
|
||||
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
|
||||
DEFAULT_MODEL = "qwen3.5:9B"
|
||||
DEFAULT_TIMEOUT = 300
|
||||
DEFAULT_NUM_CTX = 8192
|
||||
DEFAULT_NUM_PREDICT = 2048
|
||||
|
||||
EVENT_TYPES = {
|
||||
"idea",
|
||||
"option",
|
||||
"proposal",
|
||||
"objection",
|
||||
"supporting_argument",
|
||||
"clarification",
|
||||
"rejection",
|
||||
"scoped_acceptance",
|
||||
"fact",
|
||||
"technical_finding",
|
||||
}
|
||||
OUTCOME_STATUSES = {"established", "rejected", "scoped_acceptance", "tentative"}
|
||||
|
||||
|
||||
class SynthesisValidationError(ValueError):
|
||||
"""Raised when isolated semantic synthesis output is structurally invalid."""
|
||||
|
||||
|
||||
PROMPT_TEMPLATE = """You perform semantic synthesis for one already known discussion subject.
|
||||
|
||||
The subject boundary and evidence assignment are fixed and complete. Do not discover,
|
||||
split, merge, rename, or omit the subject. Do not assign evidence to another subject.
|
||||
Interpret only what the supplied evidence semantically establishes.
|
||||
|
||||
Semantic distinctions:
|
||||
- idea: mentioned possibility without stronger commitment
|
||||
- option: alternative considered without commitment
|
||||
- proposal: suggested course of action not yet established as work
|
||||
- objection: argument or concern against something; not automatically unresolved
|
||||
- rejection: an alternative is explicitly rejected
|
||||
- scoped_acceptance: accepted only for the stated test, trial, condition, or scope
|
||||
- proposal is not an action
|
||||
- no decision is not a tentative decision
|
||||
- mention is not an unresolved issue
|
||||
- an action requires explicit assignment, acceptance, commitment, or established work
|
||||
- an unresolved issue requires a concrete need explicitly left unresolved
|
||||
|
||||
Preserve explicit rejection, explicit accepted work, explicit unresolved questions,
|
||||
and all limits on an outcome. Never generalize trial acceptance into final acceptance.
|
||||
Use only supplied evidence IDs. Keep concise semantic text in the evidence language.
|
||||
|
||||
Return exactly one JSON object. Always include these fields:
|
||||
{{
|
||||
"schema_version": "experimental-semantic-synthesis-v1",
|
||||
"subject_id": "copy the supplied subject_id exactly",
|
||||
"subject": "copy the supplied subject exactly",
|
||||
"events": [
|
||||
{{
|
||||
"type": "idea|option|proposal|objection|supporting_argument|clarification|rejection|scoped_acceptance|fact|technical_finding",
|
||||
"text": "supported semantic event",
|
||||
"evidence_ids": ["e1"]
|
||||
}}
|
||||
],
|
||||
"actions": [
|
||||
{{
|
||||
"text": "established action",
|
||||
"responsible": null,
|
||||
"due": null,
|
||||
"evidence_ids": ["e2"]
|
||||
}}
|
||||
],
|
||||
"unresolved_issues": [
|
||||
{{
|
||||
"text": "explicitly unresolved issue",
|
||||
"evidence_ids": ["e3"]
|
||||
}}
|
||||
]
|
||||
}}
|
||||
|
||||
The three arrays are structurally required; use [] when none exist.
|
||||
Add "outcome" only when an outcome was actually established:
|
||||
{{
|
||||
"status": "established|rejected|scoped_acceptance|tentative",
|
||||
"text": "what was actually established",
|
||||
"scope": "the exact scope, condition, or limit",
|
||||
"evidence_ids": ["e2"]
|
||||
}}
|
||||
Omit outcome completely when there is none. Never use null for outcome. Never use the
|
||||
string "null"; use JSON null only for unknown responsible or due values.
|
||||
|
||||
Fixed Gold input:
|
||||
{input_json}
|
||||
"""
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Run the isolated semantic-synthesis Gold experiment."
|
||||
)
|
||||
parser.add_argument("fixture", type=Path, help="Fixed-subject Gold bundle JSON.")
|
||||
parser.add_argument("-o", "--output", type=Path, required=True)
|
||||
parser.add_argument("--model", default=DEFAULT_MODEL)
|
||||
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
|
||||
parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT)
|
||||
parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX)
|
||||
parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def _exact_keys(
|
||||
value: dict[str, Any], required: set[str], optional: set[str], location: str
|
||||
) -> None:
|
||||
missing = required - value.keys()
|
||||
unknown = value.keys() - required - optional
|
||||
if missing:
|
||||
raise SynthesisValidationError(
|
||||
f"{location} missing required keys: {sorted(missing)}"
|
||||
)
|
||||
if unknown:
|
||||
raise SynthesisValidationError(
|
||||
f"{location} has unknown keys: {sorted(unknown)}"
|
||||
)
|
||||
|
||||
|
||||
def _text(value: Any, location: str) -> str:
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise SynthesisValidationError(f"{location} must be a non-empty string")
|
||||
return value.strip()
|
||||
|
||||
|
||||
def validate_bundle(case: Any) -> dict[str, Any]:
|
||||
if not isinstance(case, dict):
|
||||
raise SynthesisValidationError("case must be an object")
|
||||
_exact_keys(
|
||||
case,
|
||||
{
|
||||
"case_id",
|
||||
"description",
|
||||
"subject_id",
|
||||
"subject",
|
||||
"evidence",
|
||||
"allowed_responsible",
|
||||
"expected",
|
||||
},
|
||||
set(),
|
||||
"case",
|
||||
)
|
||||
_text(case["case_id"], "case.case_id")
|
||||
_text(case["description"], "case.description")
|
||||
_text(case["subject_id"], "case.subject_id")
|
||||
_text(case["subject"], "case.subject")
|
||||
evidence = case["evidence"]
|
||||
if not isinstance(evidence, list) or not evidence:
|
||||
raise SynthesisValidationError("case.evidence must be a non-empty list")
|
||||
seen: set[str] = set()
|
||||
for index, item in enumerate(evidence):
|
||||
location = f"case.evidence[{index}]"
|
||||
if not isinstance(item, dict):
|
||||
raise SynthesisValidationError(f"{location} must be an object")
|
||||
_exact_keys(item, {"evidence_id", "text"}, set(), location)
|
||||
evidence_id = _text(item["evidence_id"], f"{location}.evidence_id")
|
||||
if evidence_id in seen:
|
||||
raise SynthesisValidationError(f"duplicate evidence ID: {evidence_id}")
|
||||
seen.add(evidence_id)
|
||||
_text(item["text"], f"{location}.text")
|
||||
allowed = case["allowed_responsible"]
|
||||
if not isinstance(allowed, list) or any(
|
||||
not isinstance(value, str) or not value.strip() for value in allowed
|
||||
):
|
||||
raise SynthesisValidationError(
|
||||
"case.allowed_responsible must be a list of non-empty strings"
|
||||
)
|
||||
if len(set(allowed)) != len(allowed):
|
||||
raise SynthesisValidationError("case.allowed_responsible contains duplicates")
|
||||
if not isinstance(case["expected"], dict):
|
||||
raise SynthesisValidationError("case.expected must be an object")
|
||||
return case
|
||||
|
||||
|
||||
def _evidence_ids(value: Any, location: str, known: set[str]) -> list[str]:
|
||||
if not isinstance(value, list) or not value:
|
||||
raise SynthesisValidationError(f"{location} must be a non-empty list")
|
||||
result: list[str] = []
|
||||
for index, evidence_id in enumerate(value):
|
||||
evidence_id = _text(evidence_id, f"{location}[{index}]")
|
||||
if evidence_id not in known:
|
||||
raise SynthesisValidationError(
|
||||
f"{location}[{index}] references unknown evidence ID: {evidence_id}"
|
||||
)
|
||||
if evidence_id in result:
|
||||
raise SynthesisValidationError(
|
||||
f"{location} contains duplicate evidence ID: {evidence_id}"
|
||||
)
|
||||
result.append(evidence_id)
|
||||
return result
|
||||
|
||||
|
||||
def _nullable_text(value: Any, location: str) -> str | None:
|
||||
if value is None:
|
||||
return None
|
||||
result = _text(value, location)
|
||||
if result.casefold() == "null":
|
||||
raise SynthesisValidationError(
|
||||
f"{location} must use JSON null, not the string 'null'"
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
def validate_synthesis(data: Any, case: dict[str, Any]) -> dict[str, Any]:
|
||||
validate_bundle(case)
|
||||
if not isinstance(data, dict):
|
||||
raise SynthesisValidationError("output must be an object")
|
||||
_exact_keys(
|
||||
data,
|
||||
{
|
||||
"schema_version",
|
||||
"subject_id",
|
||||
"subject",
|
||||
"events",
|
||||
"actions",
|
||||
"unresolved_issues",
|
||||
},
|
||||
{"outcome"},
|
||||
"output",
|
||||
)
|
||||
if data["schema_version"] != SCHEMA_VERSION:
|
||||
raise SynthesisValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
|
||||
if data["subject_id"] != case["subject_id"]:
|
||||
raise SynthesisValidationError("model changed fixed subject_id")
|
||||
if data["subject"] != case["subject"]:
|
||||
raise SynthesisValidationError("model changed fixed subject")
|
||||
|
||||
known = {item["evidence_id"] for item in case["evidence"]}
|
||||
events = data["events"]
|
||||
if not isinstance(events, list):
|
||||
raise SynthesisValidationError("output.events must be an array")
|
||||
for index, event in enumerate(events):
|
||||
location = f"output.events[{index}]"
|
||||
if not isinstance(event, dict):
|
||||
raise SynthesisValidationError(f"{location} must be an object")
|
||||
_exact_keys(event, {"type", "text", "evidence_ids"}, set(), location)
|
||||
if event["type"] not in EVENT_TYPES:
|
||||
raise SynthesisValidationError(f"{location}.type is invalid")
|
||||
_text(event["text"], f"{location}.text")
|
||||
_evidence_ids(event["evidence_ids"], f"{location}.evidence_ids", known)
|
||||
|
||||
if "outcome" in data:
|
||||
outcome = data["outcome"]
|
||||
if not isinstance(outcome, dict):
|
||||
raise SynthesisValidationError(
|
||||
"output.outcome must be an object when present; omit it when absent"
|
||||
)
|
||||
_exact_keys(
|
||||
outcome, {"status", "text", "scope", "evidence_ids"}, set(), "output.outcome"
|
||||
)
|
||||
if outcome["status"] not in OUTCOME_STATUSES:
|
||||
raise SynthesisValidationError("output.outcome.status is invalid")
|
||||
_text(outcome["text"], "output.outcome.text")
|
||||
_text(outcome["scope"], "output.outcome.scope")
|
||||
_evidence_ids(outcome["evidence_ids"], "output.outcome.evidence_ids", known)
|
||||
|
||||
actions = data["actions"]
|
||||
if not isinstance(actions, list):
|
||||
raise SynthesisValidationError("output.actions must be an array")
|
||||
allowed = set(case["allowed_responsible"])
|
||||
for index, action in enumerate(actions):
|
||||
location = f"output.actions[{index}]"
|
||||
if not isinstance(action, dict):
|
||||
raise SynthesisValidationError(f"{location} must be an object")
|
||||
_exact_keys(
|
||||
action,
|
||||
{"text", "responsible", "due", "evidence_ids"},
|
||||
set(),
|
||||
location,
|
||||
)
|
||||
_text(action["text"], f"{location}.text")
|
||||
responsible = _nullable_text(action["responsible"], f"{location}.responsible")
|
||||
if responsible is not None and responsible not in allowed:
|
||||
raise SynthesisValidationError(
|
||||
f"{location}.responsible is not allowed: {responsible}"
|
||||
)
|
||||
_nullable_text(action["due"], f"{location}.due")
|
||||
_evidence_ids(action["evidence_ids"], f"{location}.evidence_ids", known)
|
||||
|
||||
issues = data["unresolved_issues"]
|
||||
if not isinstance(issues, list):
|
||||
raise SynthesisValidationError("output.unresolved_issues must be an array")
|
||||
for index, issue in enumerate(issues):
|
||||
location = f"output.unresolved_issues[{index}]"
|
||||
if not isinstance(issue, dict):
|
||||
raise SynthesisValidationError(f"{location} must be an object")
|
||||
_exact_keys(issue, {"text", "evidence_ids"}, set(), location)
|
||||
_text(issue["text"], f"{location}.text")
|
||||
_evidence_ids(issue["evidence_ids"], f"{location}.evidence_ids", known)
|
||||
return data
|
||||
|
||||
|
||||
def build_prompt(case: dict[str, Any]) -> str:
|
||||
validate_bundle(case)
|
||||
model_input = {
|
||||
"subject_id": case["subject_id"],
|
||||
"subject": case["subject"],
|
||||
"evidence": case["evidence"],
|
||||
}
|
||||
return PROMPT_TEMPLATE.format(
|
||||
input_json=json.dumps(model_input, ensure_ascii=False, indent=2)
|
||||
)
|
||||
|
||||
|
||||
def parse_model_json(raw_text: str) -> dict[str, Any]:
|
||||
data = json.loads(raw_text)
|
||||
if not isinstance(data, dict):
|
||||
raise SynthesisValidationError("model response JSON must be an object")
|
||||
return data
|
||||
|
||||
|
||||
def build_ollama_payload(
|
||||
model: str, prompt: str, num_ctx: int, num_predict: int
|
||||
) -> dict[str, Any]:
|
||||
return {
|
||||
"model": model,
|
||||
"prompt": prompt,
|
||||
"think": False,
|
||||
"stream": False,
|
||||
"format": "json",
|
||||
"options": {
|
||||
"temperature": 0,
|
||||
"num_ctx": num_ctx,
|
||||
"num_predict": num_predict,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def call_ollama(
|
||||
endpoint: str,
|
||||
model: str,
|
||||
prompt: str,
|
||||
timeout: int,
|
||||
num_ctx: int,
|
||||
num_predict: int,
|
||||
) -> tuple[str, dict[str, Any]]:
|
||||
payload = build_ollama_payload(model, prompt, num_ctx, num_predict)
|
||||
started = time.perf_counter()
|
||||
response = requests.post(endpoint, json=payload, timeout=timeout)
|
||||
elapsed = time.perf_counter() - started
|
||||
response.raise_for_status()
|
||||
body = response.json()
|
||||
if not isinstance(body, dict):
|
||||
raise ValueError("Ollama response must be an object")
|
||||
raw_text = body.get("response")
|
||||
if not isinstance(raw_text, str) or not raw_text.strip():
|
||||
raise ValueError("Ollama returned no usable response text")
|
||||
metadata = {
|
||||
"model": body.get("model", model),
|
||||
"elapsed_seconds": round(elapsed, 3),
|
||||
"total_duration_ns": body.get("total_duration"),
|
||||
"load_duration_ns": body.get("load_duration"),
|
||||
"prompt_eval_count": body.get("prompt_eval_count"),
|
||||
"prompt_eval_duration_ns": body.get("prompt_eval_duration"),
|
||||
"eval_count": body.get("eval_count"),
|
||||
"eval_duration_ns": body.get("eval_duration"),
|
||||
"configuration": {
|
||||
"temperature": 0,
|
||||
"think": False,
|
||||
"num_ctx": num_ctx,
|
||||
"num_predict": num_predict,
|
||||
},
|
||||
}
|
||||
return raw_text.strip(), metadata
|
||||
|
||||
|
||||
def _contains(text: str, terms: list[str]) -> bool:
|
||||
folded = text.casefold()
|
||||
return any(term.casefold() in folded for term in terms)
|
||||
|
||||
|
||||
def _refs_cover(items: list[dict[str, Any]], expected: list[str]) -> bool:
|
||||
actual = {
|
||||
evidence_id
|
||||
for item in items
|
||||
for evidence_id in item.get("evidence_ids", [])
|
||||
}
|
||||
return set(expected).issubset(actual)
|
||||
|
||||
|
||||
def evaluate_synthesis(data: dict[str, Any], expected: dict[str, Any]) -> dict[str, Any]:
|
||||
checks: list[dict[str, Any]] = []
|
||||
|
||||
def add(name: str, passed: bool, critical: bool = False) -> None:
|
||||
checks.append({"name": name, "passed": passed, "critical": critical})
|
||||
|
||||
events = data["events"]
|
||||
event_types = [item["type"] for item in events]
|
||||
for event_type, minimum in expected.get("event_type_minimums", {}).items():
|
||||
add(f"event:{event_type}", event_types.count(event_type) >= minimum)
|
||||
allowed_types = set(expected.get("allowed_event_types", EVENT_TYPES))
|
||||
add("no_unexpected_event_types", set(event_types).issubset(allowed_types))
|
||||
add(
|
||||
"event_evidence",
|
||||
_refs_cover(events, expected.get("event_evidence_ids", [])),
|
||||
)
|
||||
|
||||
outcome_expected = expected["outcome"]
|
||||
outcome = data.get("outcome")
|
||||
add(
|
||||
"outcome_presence",
|
||||
(outcome is not None) == outcome_expected["required"],
|
||||
critical=True,
|
||||
)
|
||||
if outcome_expected["required"] and outcome is not None:
|
||||
add("outcome_status", outcome["status"] in outcome_expected["statuses"])
|
||||
combined = f"{outcome['text']} {outcome['scope']}"
|
||||
add("outcome_meaning", _contains(combined, outcome_expected["terms"]))
|
||||
add(
|
||||
"outcome_scope",
|
||||
_contains(combined, outcome_expected["scope_terms"]),
|
||||
critical=True,
|
||||
)
|
||||
add(
|
||||
"outcome_evidence",
|
||||
set(outcome_expected["evidence_ids"]).issubset(outcome["evidence_ids"]),
|
||||
critical=True,
|
||||
)
|
||||
|
||||
actions = data["actions"]
|
||||
expected_actions = expected["actions"]
|
||||
add(
|
||||
"action_count",
|
||||
len(actions) == expected_actions["count"],
|
||||
critical=True,
|
||||
)
|
||||
if expected_actions["count"] and actions:
|
||||
action_text = " ".join(item["text"] for item in actions)
|
||||
add("action_meaning", _contains(action_text, expected_actions["terms"]))
|
||||
if "responsible" in expected_actions:
|
||||
add(
|
||||
"action_responsible",
|
||||
any(item["responsible"] == expected_actions["responsible"] for item in actions),
|
||||
critical=True,
|
||||
)
|
||||
if expected_actions.get("due_terms"):
|
||||
due_text = " ".join(str(item["due"] or "") for item in actions)
|
||||
add("action_due", _contains(due_text, expected_actions["due_terms"]))
|
||||
add(
|
||||
"action_evidence",
|
||||
_refs_cover(actions, expected_actions["evidence_ids"]),
|
||||
critical=True,
|
||||
)
|
||||
|
||||
issues = data["unresolved_issues"]
|
||||
expected_issues = expected["unresolved_issues"]
|
||||
add(
|
||||
"unresolved_count",
|
||||
len(issues) == expected_issues["count"],
|
||||
critical=True,
|
||||
)
|
||||
if expected_issues["count"] and issues:
|
||||
issue_text = " ".join(item["text"] for item in issues)
|
||||
add("unresolved_meaning", _contains(issue_text, expected_issues["terms"]))
|
||||
add(
|
||||
"unresolved_evidence",
|
||||
_refs_cover(issues, expected_issues["evidence_ids"]),
|
||||
critical=True,
|
||||
)
|
||||
|
||||
passed = sum(item["passed"] for item in checks)
|
||||
critical_failures = [
|
||||
item["name"] for item in checks if item["critical"] and not item["passed"]
|
||||
]
|
||||
ratio = passed / len(checks)
|
||||
if ratio == 1:
|
||||
verdict = "PASS"
|
||||
elif ratio >= 0.7 and not critical_failures:
|
||||
verdict = "PARTIAL"
|
||||
else:
|
||||
verdict = "FAIL"
|
||||
failed = [item["name"] for item in checks if not item["passed"]]
|
||||
return {
|
||||
"verdict": verdict,
|
||||
"reason": "All semantic checks passed." if not failed else "Failed: " + ", ".join(failed),
|
||||
"passed_checks": passed,
|
||||
"check_count": len(checks),
|
||||
"critical_failures": critical_failures,
|
||||
"checks": checks,
|
||||
}
|
||||
|
||||
|
||||
def load_fixture(path: Path) -> list[dict[str, Any]]:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
if not isinstance(data, dict) or set(data) != {"cases"}:
|
||||
raise SynthesisValidationError("fixture must contain exactly a cases list")
|
||||
cases = data["cases"]
|
||||
if not isinstance(cases, list) or not cases:
|
||||
raise SynthesisValidationError("fixture cases must be a non-empty list")
|
||||
seen: set[str] = set()
|
||||
for case in cases:
|
||||
validate_bundle(case)
|
||||
if case["case_id"] in seen:
|
||||
raise SynthesisValidationError(f"duplicate case ID: {case['case_id']}")
|
||||
seen.add(case["case_id"])
|
||||
return cases
|
||||
|
||||
|
||||
def run_case(
|
||||
case: dict[str, Any],
|
||||
output_root: Path,
|
||||
endpoint: str,
|
||||
model: str,
|
||||
timeout: int,
|
||||
num_ctx: int,
|
||||
num_predict: int,
|
||||
) -> dict[str, Any]:
|
||||
case_dir = output_root / case["case_id"]
|
||||
case_dir.mkdir(parents=True, exist_ok=False)
|
||||
gold_input = {
|
||||
"case_id": case["case_id"],
|
||||
"description": case["description"],
|
||||
"subject_id": case["subject_id"],
|
||||
"subject": case["subject"],
|
||||
"evidence": case["evidence"],
|
||||
}
|
||||
(case_dir / "gold_input.json").write_text(
|
||||
json.dumps(gold_input, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
prompt = build_prompt(case)
|
||||
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
raw_text, metadata = call_ollama(
|
||||
endpoint, model, prompt, timeout, num_ctx, num_predict
|
||||
)
|
||||
(case_dir / "raw_model_response.txt").write_text(raw_text + "\n", encoding="utf-8")
|
||||
(case_dir / "ollama_metadata.json").write_text(
|
||||
json.dumps(metadata, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
parsed = parse_model_json(raw_text)
|
||||
(case_dir / "parsed_response.json").write_text(
|
||||
json.dumps(parsed, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
validated = validate_synthesis(parsed, case)
|
||||
validation = {"valid": True, "error": None}
|
||||
evaluation = evaluate_synthesis(validated, case["expected"])
|
||||
except requests.RequestException:
|
||||
raise
|
||||
except (json.JSONDecodeError, SynthesisValidationError, ValueError) as exc:
|
||||
validation = {
|
||||
"valid": False,
|
||||
"error_type": type(exc).__name__,
|
||||
"error": str(exc),
|
||||
}
|
||||
evaluation = {
|
||||
"verdict": "FAIL",
|
||||
"reason": f"Schema validation failed: {exc}",
|
||||
"passed_checks": 0,
|
||||
"check_count": 0,
|
||||
"critical_failures": ["schema_validation"],
|
||||
"checks": [],
|
||||
}
|
||||
(case_dir / "validation_result.json").write_text(
|
||||
json.dumps(validation, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
result = {
|
||||
"case_id": case["case_id"],
|
||||
"description": case["description"],
|
||||
**evaluation,
|
||||
"elapsed_seconds": round(time.perf_counter() - started, 3),
|
||||
}
|
||||
(case_dir / "evaluation_result.json").write_text(
|
||||
json.dumps(result, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
|
||||
cases = load_fixture(args.fixture)
|
||||
args.output.mkdir(parents=True, exist_ok=False)
|
||||
started = time.perf_counter()
|
||||
results: list[dict[str, Any]] = []
|
||||
for index, case in enumerate(cases, start=1):
|
||||
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
|
||||
results.append(
|
||||
run_case(
|
||||
case,
|
||||
args.output,
|
||||
args.endpoint,
|
||||
args.model,
|
||||
args.timeout,
|
||||
args.num_ctx,
|
||||
args.num_predict,
|
||||
)
|
||||
)
|
||||
summary = {
|
||||
"experiment": "semantic_synthesis_isolation",
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"model": args.model,
|
||||
"temperature": 0,
|
||||
"think": False,
|
||||
"num_ctx": args.num_ctx,
|
||||
"num_predict": args.num_predict,
|
||||
"case_count": len(cases),
|
||||
"llm_call_count": len(results),
|
||||
"runtime_seconds": round(time.perf_counter() - started, 3),
|
||||
"verdict_counts": {
|
||||
verdict: sum(result["verdict"] == verdict for result in results)
|
||||
for verdict in ("PASS", "PARTIAL", "FAIL")
|
||||
},
|
||||
"results": results,
|
||||
}
|
||||
(args.output / "summary.json").write_text(
|
||||
json.dumps(summary, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
||||
)
|
||||
return summary
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
try:
|
||||
summary = run_experiment(args)
|
||||
except (OSError, ValueError, requests.RequestException) as exc:
|
||||
print(f"Error: {exc}")
|
||||
return 1
|
||||
print(json.dumps(summary["verdict_counts"], sort_keys=True))
|
||||
print(f"Artifacts: {args.output.resolve()}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1 @@
|
||||
"""Experimental topic-oriented discussion reconstruction."""
|
||||
@@ -0,0 +1,721 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run an isolated Discussion Subject reconstruction experiment with Ollama."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
SCHEMA_VERSION = "experimental-discussion-subjects-v1"
|
||||
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
|
||||
DEFAULT_MODEL = "qwen3.5:9B"
|
||||
DEFAULT_TIMEOUT = 300
|
||||
DEFAULT_NUM_CTX = 16384
|
||||
DEFAULT_NUM_PREDICT = 4096
|
||||
|
||||
EVENT_TYPES = {
|
||||
"introduced_idea",
|
||||
"considered_option",
|
||||
"proposal",
|
||||
"supporting_argument",
|
||||
"objection",
|
||||
"clarification",
|
||||
"modification",
|
||||
"fact",
|
||||
"technical_finding",
|
||||
}
|
||||
OUTCOME_CERTAINTIES = {"established", "tentative", "conditional", "rejected"}
|
||||
IDENTIFIER_RE = re.compile(r"^[a-z][a-z0-9_]*$")
|
||||
|
||||
|
||||
class ReconstructionValidationError(ValueError):
|
||||
"""Raised when experimental reconstruction output violates the schema."""
|
||||
|
||||
|
||||
PROMPT_TEMPLATE = """You reconstruct discussion subjects from meeting evidence.
|
||||
|
||||
This is semantic reconstruction, not protocol writing and not flat category extraction.
|
||||
Group evidence by what participants are actually discussing. For each subject, record
|
||||
only supported discourse events and, when present, the actual outcome, resulting
|
||||
actions, and genuinely unresolved issues.
|
||||
|
||||
Important distinctions:
|
||||
- discussed is not necessarily proposed
|
||||
- proposed is not necessarily preferred or accepted
|
||||
- preferred is not accepted
|
||||
- accepted for a trial is not accepted as a final solution
|
||||
- mentioned is not an unresolved question
|
||||
- an outcome must preserve its scope, conditions, polarity, and uncertainty
|
||||
- do not infer responsibility from mention, expertise, adjacency, or likely role
|
||||
- do not invent missing stages or emit empty optional structures
|
||||
|
||||
Evidence discipline:
|
||||
- Use only the supplied evidence IDs in evidence_refs.
|
||||
- Every subject, event, outcome, action, and unresolved issue needs at least one
|
||||
evidence reference.
|
||||
- Keep statements concise; do not copy long evidence passages.
|
||||
- A subject may consist only of one introduced idea.
|
||||
|
||||
Return one JSON object with exactly:
|
||||
{{
|
||||
"schema_version": "experimental-discussion-subjects-v1",
|
||||
"subjects": [
|
||||
{{
|
||||
"subject_id": "subject_1",
|
||||
"title": "concise discussion subject",
|
||||
"evidence_refs": ["e1"],
|
||||
"development": [
|
||||
{{
|
||||
"event_id": "event_1",
|
||||
"type": "introduced_idea|considered_option|proposal|supporting_argument|objection|clarification|modification|fact|technical_finding",
|
||||
"text": "what happened in the discussion",
|
||||
"evidence_refs": ["e1"]
|
||||
}}
|
||||
],
|
||||
"outcome": {{
|
||||
"text": "only what was established",
|
||||
"scope": "explicit limit or full scope of the outcome",
|
||||
"certainty": "established|tentative|conditional|rejected",
|
||||
"evidence_refs": ["e2"]
|
||||
}},
|
||||
"actions": [
|
||||
{{
|
||||
"action_id": "action_1",
|
||||
"text": "established work only",
|
||||
"responsible": "explicitly supported name or null",
|
||||
"deadline": "explicitly supported deadline or null",
|
||||
"evidence_refs": ["e3"]
|
||||
}}
|
||||
],
|
||||
"unresolved_issues": [
|
||||
{{
|
||||
"issue_id": "issue_1",
|
||||
"text": "concrete unresolved issue",
|
||||
"evidence_refs": ["e4"]
|
||||
}}
|
||||
]
|
||||
}}
|
||||
]
|
||||
}}
|
||||
|
||||
Only subject_id, title, evidence_refs are required for each subject. Omit
|
||||
development, outcome, actions, or unresolved_issues when absent. Never emit null
|
||||
or an empty optional list/object.
|
||||
|
||||
Case ID: {case_id}
|
||||
Evidence units:
|
||||
{evidence_json}
|
||||
"""
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Run the isolated topic-reconstruction Gold experiment."
|
||||
)
|
||||
parser.add_argument("fixture", type=Path, help="Focused Gold cases JSON.")
|
||||
parser.add_argument("-o", "--output", type=Path, required=True)
|
||||
parser.add_argument("--model", default=DEFAULT_MODEL)
|
||||
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
|
||||
parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT)
|
||||
parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX)
|
||||
parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT)
|
||||
parser.add_argument(
|
||||
"--case", action="append", dest="case_ids", help="Run only this case ID."
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def _expect_exact_keys(
|
||||
value: dict[str, Any], required: set[str], optional: set[str], location: str
|
||||
) -> None:
|
||||
missing = required - value.keys()
|
||||
unknown = value.keys() - required - optional
|
||||
if missing:
|
||||
raise ReconstructionValidationError(
|
||||
f"{location} missing required keys: {sorted(missing)}"
|
||||
)
|
||||
if unknown:
|
||||
raise ReconstructionValidationError(
|
||||
f"{location} has unknown keys: {sorted(unknown)}"
|
||||
)
|
||||
|
||||
|
||||
def _nonempty_text(value: Any, location: str) -> str:
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise ReconstructionValidationError(f"{location} must be a non-empty string")
|
||||
return value.strip()
|
||||
|
||||
|
||||
def _identifier(value: Any, location: str, seen: set[str]) -> str:
|
||||
text = _nonempty_text(value, location)
|
||||
if not IDENTIFIER_RE.fullmatch(text):
|
||||
raise ReconstructionValidationError(f"{location} is not a valid identifier")
|
||||
if text in seen:
|
||||
raise ReconstructionValidationError(f"duplicate identifier: {text}")
|
||||
seen.add(text)
|
||||
return text
|
||||
|
||||
|
||||
def _nullable_text(value: Any, location: str) -> str | None:
|
||||
if value is None:
|
||||
return None
|
||||
text = _nonempty_text(value, location)
|
||||
if text.casefold() == "null":
|
||||
raise ReconstructionValidationError(
|
||||
f"{location} must use JSON null, not the string 'null'"
|
||||
)
|
||||
return text
|
||||
|
||||
|
||||
def _evidence_refs(value: Any, location: str, known: set[str]) -> list[str]:
|
||||
if not isinstance(value, list) or not value:
|
||||
raise ReconstructionValidationError(f"{location} must be a non-empty list")
|
||||
refs: list[str] = []
|
||||
for index, ref in enumerate(value):
|
||||
ref = _nonempty_text(ref, f"{location}[{index}]")
|
||||
if ref not in known:
|
||||
raise ReconstructionValidationError(
|
||||
f"{location}[{index}] references unknown evidence ID: {ref}"
|
||||
)
|
||||
if ref in refs:
|
||||
raise ReconstructionValidationError(
|
||||
f"{location} contains duplicate evidence reference: {ref}"
|
||||
)
|
||||
refs.append(ref)
|
||||
return refs
|
||||
|
||||
|
||||
def validate_evidence_units(evidence_units: Any) -> set[str]:
|
||||
if not isinstance(evidence_units, list) or not evidence_units:
|
||||
raise ReconstructionValidationError("evidence_units must be a non-empty list")
|
||||
known: set[str] = set()
|
||||
for index, unit in enumerate(evidence_units):
|
||||
location = f"evidence_units[{index}]"
|
||||
if not isinstance(unit, dict):
|
||||
raise ReconstructionValidationError(f"{location} must be an object")
|
||||
_expect_exact_keys(unit, {"evidence_id", "text"}, set(), location)
|
||||
evidence_id = _nonempty_text(unit["evidence_id"], f"{location}.evidence_id")
|
||||
if evidence_id in known:
|
||||
raise ReconstructionValidationError(
|
||||
f"duplicate input evidence identifier: {evidence_id}"
|
||||
)
|
||||
known.add(evidence_id)
|
||||
_nonempty_text(unit["text"], f"{location}.text")
|
||||
return known
|
||||
|
||||
|
||||
def validate_reconstruction(data: Any, evidence_units: Any) -> dict[str, Any]:
|
||||
known = validate_evidence_units(evidence_units)
|
||||
if not isinstance(data, dict):
|
||||
raise ReconstructionValidationError("model output must be an object")
|
||||
_expect_exact_keys(data, {"schema_version", "subjects"}, set(), "output")
|
||||
if data["schema_version"] != SCHEMA_VERSION:
|
||||
raise ReconstructionValidationError(
|
||||
f"schema_version must be {SCHEMA_VERSION!r}"
|
||||
)
|
||||
subjects = data["subjects"]
|
||||
if not isinstance(subjects, list) or not subjects:
|
||||
raise ReconstructionValidationError("subjects must be a non-empty list")
|
||||
|
||||
seen: set[str] = set()
|
||||
for subject_index, subject in enumerate(subjects):
|
||||
location = f"subjects[{subject_index}]"
|
||||
if not isinstance(subject, dict):
|
||||
raise ReconstructionValidationError(f"{location} must be an object")
|
||||
_expect_exact_keys(
|
||||
subject,
|
||||
{"subject_id", "title", "evidence_refs"},
|
||||
{"development", "outcome", "actions", "unresolved_issues"},
|
||||
location,
|
||||
)
|
||||
_identifier(subject["subject_id"], f"{location}.subject_id", seen)
|
||||
_nonempty_text(subject["title"], f"{location}.title")
|
||||
_evidence_refs(subject["evidence_refs"], f"{location}.evidence_refs", known)
|
||||
|
||||
if "development" in subject:
|
||||
events = subject["development"]
|
||||
if not isinstance(events, list) or not events:
|
||||
raise ReconstructionValidationError(
|
||||
f"{location}.development must be a non-empty list when present"
|
||||
)
|
||||
for event_index, event in enumerate(events):
|
||||
event_location = f"{location}.development[{event_index}]"
|
||||
if not isinstance(event, dict):
|
||||
raise ReconstructionValidationError(
|
||||
f"{event_location} must be an object"
|
||||
)
|
||||
_expect_exact_keys(
|
||||
event,
|
||||
{"event_id", "type", "text", "evidence_refs"},
|
||||
set(),
|
||||
event_location,
|
||||
)
|
||||
_identifier(event["event_id"], f"{event_location}.event_id", seen)
|
||||
if event["type"] not in EVENT_TYPES:
|
||||
raise ReconstructionValidationError(
|
||||
f"{event_location}.type is invalid: {event['type']!r}"
|
||||
)
|
||||
_nonempty_text(event["text"], f"{event_location}.text")
|
||||
_evidence_refs(
|
||||
event["evidence_refs"], f"{event_location}.evidence_refs", known
|
||||
)
|
||||
|
||||
if "outcome" in subject:
|
||||
outcome = subject["outcome"]
|
||||
outcome_location = f"{location}.outcome"
|
||||
if not isinstance(outcome, dict):
|
||||
raise ReconstructionValidationError(
|
||||
f"{outcome_location} must be a non-empty object when present"
|
||||
)
|
||||
_expect_exact_keys(
|
||||
outcome,
|
||||
{"text", "scope", "certainty", "evidence_refs"},
|
||||
set(),
|
||||
outcome_location,
|
||||
)
|
||||
_nonempty_text(outcome["text"], f"{outcome_location}.text")
|
||||
_nonempty_text(outcome["scope"], f"{outcome_location}.scope")
|
||||
if outcome["certainty"] not in OUTCOME_CERTAINTIES:
|
||||
raise ReconstructionValidationError(
|
||||
f"{outcome_location}.certainty is invalid: {outcome['certainty']!r}"
|
||||
)
|
||||
_evidence_refs(
|
||||
outcome["evidence_refs"], f"{outcome_location}.evidence_refs", known
|
||||
)
|
||||
|
||||
if "actions" in subject:
|
||||
actions = subject["actions"]
|
||||
if not isinstance(actions, list) or not actions:
|
||||
raise ReconstructionValidationError(
|
||||
f"{location}.actions must be a non-empty list when present"
|
||||
)
|
||||
for action_index, action in enumerate(actions):
|
||||
action_location = f"{location}.actions[{action_index}]"
|
||||
if not isinstance(action, dict):
|
||||
raise ReconstructionValidationError(
|
||||
f"{action_location} must be an object"
|
||||
)
|
||||
_expect_exact_keys(
|
||||
action,
|
||||
{"action_id", "text", "responsible", "deadline", "evidence_refs"},
|
||||
set(),
|
||||
action_location,
|
||||
)
|
||||
_identifier(action["action_id"], f"{action_location}.action_id", seen)
|
||||
_nonempty_text(action["text"], f"{action_location}.text")
|
||||
for field in ("responsible", "deadline"):
|
||||
_nullable_text(action[field], f"{action_location}.{field}")
|
||||
_evidence_refs(
|
||||
action["evidence_refs"], f"{action_location}.evidence_refs", known
|
||||
)
|
||||
|
||||
if "unresolved_issues" in subject:
|
||||
issues = subject["unresolved_issues"]
|
||||
if not isinstance(issues, list) or not issues:
|
||||
raise ReconstructionValidationError(
|
||||
f"{location}.unresolved_issues must be a non-empty list when present"
|
||||
)
|
||||
for issue_index, issue in enumerate(issues):
|
||||
issue_location = f"{location}.unresolved_issues[{issue_index}]"
|
||||
if not isinstance(issue, dict):
|
||||
raise ReconstructionValidationError(
|
||||
f"{issue_location} must be an object"
|
||||
)
|
||||
_expect_exact_keys(
|
||||
issue,
|
||||
{"issue_id", "text", "evidence_refs"},
|
||||
set(),
|
||||
issue_location,
|
||||
)
|
||||
_identifier(issue["issue_id"], f"{issue_location}.issue_id", seen)
|
||||
_nonempty_text(issue["text"], f"{issue_location}.text")
|
||||
_evidence_refs(
|
||||
issue["evidence_refs"], f"{issue_location}.evidence_refs", known
|
||||
)
|
||||
|
||||
return data
|
||||
|
||||
|
||||
def build_prompt(case: dict[str, Any]) -> str:
|
||||
evidence_units = case["evidence_units"]
|
||||
validate_evidence_units(evidence_units)
|
||||
return PROMPT_TEMPLATE.format(
|
||||
case_id=case["case_id"],
|
||||
evidence_json=json.dumps(evidence_units, ensure_ascii=False, indent=2),
|
||||
)
|
||||
|
||||
|
||||
def parse_model_json(raw_text: str) -> dict[str, Any]:
|
||||
data = json.loads(raw_text)
|
||||
if not isinstance(data, dict):
|
||||
raise ReconstructionValidationError("model response JSON must be an object")
|
||||
return data
|
||||
|
||||
|
||||
def build_ollama_payload(
|
||||
model: str, prompt: str, num_ctx: int, num_predict: int
|
||||
) -> dict[str, Any]:
|
||||
return {
|
||||
"model": model,
|
||||
"prompt": prompt,
|
||||
"think": False,
|
||||
"stream": False,
|
||||
"format": "json",
|
||||
"options": {
|
||||
"temperature": 0,
|
||||
"num_ctx": num_ctx,
|
||||
"num_predict": num_predict,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def call_ollama(
|
||||
endpoint: str,
|
||||
model: str,
|
||||
prompt: str,
|
||||
timeout: int,
|
||||
num_ctx: int,
|
||||
num_predict: int,
|
||||
) -> tuple[str, dict[str, Any]]:
|
||||
payload = build_ollama_payload(model, prompt, num_ctx, num_predict)
|
||||
started = time.perf_counter()
|
||||
response = requests.post(endpoint, json=payload, timeout=timeout)
|
||||
elapsed = time.perf_counter() - started
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
if not isinstance(data, dict):
|
||||
raise ValueError("Ollama response must be a JSON object")
|
||||
raw_text = data.get("response")
|
||||
if not isinstance(raw_text, str) or not raw_text.strip():
|
||||
raise ValueError("Ollama returned no usable response text")
|
||||
metadata = {
|
||||
"model": data.get("model", model),
|
||||
"elapsed_seconds": round(elapsed, 3),
|
||||
"total_duration_ns": data.get("total_duration"),
|
||||
"load_duration_ns": data.get("load_duration"),
|
||||
"prompt_eval_count": data.get("prompt_eval_count"),
|
||||
"prompt_eval_duration_ns": data.get("prompt_eval_duration"),
|
||||
"eval_count": data.get("eval_count"),
|
||||
"eval_duration_ns": data.get("eval_duration"),
|
||||
"configuration": {
|
||||
"temperature": 0,
|
||||
"think": False,
|
||||
"num_ctx": num_ctx,
|
||||
"num_predict": num_predict,
|
||||
},
|
||||
}
|
||||
return raw_text.strip(), metadata
|
||||
|
||||
|
||||
def _all_text(subjects: list[dict[str, Any]]) -> str:
|
||||
parts: list[str] = []
|
||||
for subject in subjects:
|
||||
parts.append(subject["title"])
|
||||
for event in subject.get("development", []):
|
||||
parts.append(event["text"])
|
||||
outcome = subject.get("outcome")
|
||||
if outcome:
|
||||
parts.extend((outcome["text"], outcome["scope"]))
|
||||
for action in subject.get("actions", []):
|
||||
parts.append(action["text"])
|
||||
for issue in subject.get("unresolved_issues", []):
|
||||
parts.append(issue["text"])
|
||||
return " ".join(parts).casefold()
|
||||
|
||||
|
||||
def _contains_any(text: str, terms: list[str]) -> bool:
|
||||
return any(term.casefold() in text for term in terms)
|
||||
|
||||
|
||||
def evaluate_reconstruction(
|
||||
reconstruction: dict[str, Any], expected: dict[str, Any]
|
||||
) -> dict[str, Any]:
|
||||
subjects = reconstruction["subjects"]
|
||||
combined = _all_text(subjects)
|
||||
events = [event for subject in subjects for event in subject.get("development", [])]
|
||||
outcomes = [subject["outcome"] for subject in subjects if "outcome" in subject]
|
||||
actions = [action for subject in subjects for action in subject.get("actions", [])]
|
||||
issues = [issue for subject in subjects for issue in subject.get("unresolved_issues", [])]
|
||||
checks: list[dict[str, Any]] = []
|
||||
|
||||
def add(name: str, passed: bool, critical: bool = False) -> None:
|
||||
checks.append({"name": name, "passed": passed, "critical": critical})
|
||||
|
||||
add("subject_count", len(subjects) == expected.get("subject_count", 1))
|
||||
add("subject_identity", _contains_any(combined, expected["subject_terms"]))
|
||||
|
||||
event_types = {event["type"] for event in events}
|
||||
for event_type in expected.get("required_event_types", []):
|
||||
add(f"event_type:{event_type}", event_type in event_types)
|
||||
|
||||
expected_outcome = expected.get("outcome", {})
|
||||
outcome_required = expected_outcome.get("required", False)
|
||||
add(
|
||||
"outcome_presence",
|
||||
bool(outcomes) is outcome_required,
|
||||
critical=not outcome_required and bool(outcomes),
|
||||
)
|
||||
if outcome_required and outcomes:
|
||||
outcome_text = " ".join(
|
||||
f"{item['text']} {item['scope']}" for item in outcomes
|
||||
).casefold()
|
||||
add("outcome_meaning", _contains_any(outcome_text, expected_outcome["terms"]))
|
||||
add(
|
||||
"outcome_scope",
|
||||
_contains_any(outcome_text, expected_outcome.get("scope_terms", [])),
|
||||
critical=True,
|
||||
)
|
||||
add(
|
||||
"outcome_certainty",
|
||||
any(
|
||||
item["certainty"] in expected_outcome.get("certainties", [])
|
||||
for item in outcomes
|
||||
),
|
||||
)
|
||||
|
||||
expected_actions = expected.get("actions", {})
|
||||
minimum_actions = expected_actions.get("minimum", 0)
|
||||
add(
|
||||
"action_count",
|
||||
len(actions) >= minimum_actions if minimum_actions else not actions,
|
||||
critical=minimum_actions == 0 and bool(actions),
|
||||
)
|
||||
if minimum_actions and actions:
|
||||
action_text = " ".join(item["text"] for item in actions).casefold()
|
||||
add("action_meaning", _contains_any(action_text, expected_actions["terms"]))
|
||||
if "responsible" in expected_actions:
|
||||
add(
|
||||
"action_responsibility",
|
||||
any(
|
||||
item["responsible"] == expected_actions["responsible"]
|
||||
for item in actions
|
||||
),
|
||||
critical=True,
|
||||
)
|
||||
|
||||
expected_issues = expected.get("unresolved", {})
|
||||
minimum_issues = expected_issues.get("minimum", 0)
|
||||
add(
|
||||
"unresolved_count",
|
||||
len(issues) >= minimum_issues if minimum_issues else not issues,
|
||||
critical=minimum_issues == 0 and bool(issues),
|
||||
)
|
||||
if minimum_issues and issues:
|
||||
issue_text = " ".join(item["text"] for item in issues).casefold()
|
||||
add("unresolved_meaning", _contains_any(issue_text, expected_issues["terms"]))
|
||||
|
||||
passed = sum(check["passed"] for check in checks)
|
||||
critical_failures = [
|
||||
check["name"] for check in checks if check["critical"] and not check["passed"]
|
||||
]
|
||||
ratio = passed / len(checks)
|
||||
if ratio == 1:
|
||||
verdict = "PASS"
|
||||
elif ratio >= 0.6 and not critical_failures:
|
||||
verdict = "PARTIAL"
|
||||
else:
|
||||
verdict = "FAIL"
|
||||
failed = [check["name"] for check in checks if not check["passed"]]
|
||||
reason = "All semantic checks passed." if not failed else "Failed: " + ", ".join(failed)
|
||||
return {
|
||||
"verdict": verdict,
|
||||
"reason": reason,
|
||||
"passed_checks": passed,
|
||||
"check_count": len(checks),
|
||||
"critical_failures": critical_failures,
|
||||
"checks": checks,
|
||||
}
|
||||
|
||||
|
||||
def load_fixture(path: Path) -> list[dict[str, Any]]:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
if not isinstance(data, dict) or set(data) != {"cases"}:
|
||||
raise ValueError("fixture must contain exactly one 'cases' list")
|
||||
cases = data["cases"]
|
||||
if not isinstance(cases, list) or not cases:
|
||||
raise ValueError("fixture cases must be a non-empty list")
|
||||
seen: set[str] = set()
|
||||
for index, case in enumerate(cases):
|
||||
if not isinstance(case, dict):
|
||||
raise ValueError(f"cases[{index}] must be an object")
|
||||
required = {"case_id", "description", "evidence_units", "expected"}
|
||||
if set(case) != required:
|
||||
raise ValueError(f"cases[{index}] must contain exactly {sorted(required)}")
|
||||
case_id = _nonempty_text(case["case_id"], f"cases[{index}].case_id")
|
||||
if case_id in seen:
|
||||
raise ValueError(f"duplicate case_id: {case_id}")
|
||||
seen.add(case_id)
|
||||
_nonempty_text(case["description"], f"cases[{index}].description")
|
||||
validate_evidence_units(case["evidence_units"])
|
||||
if not isinstance(case["expected"], dict):
|
||||
raise ValueError(f"cases[{index}].expected must be an object")
|
||||
return cases
|
||||
|
||||
|
||||
def run_case(
|
||||
case: dict[str, Any],
|
||||
output_root: Path,
|
||||
endpoint: str,
|
||||
model: str,
|
||||
timeout: int,
|
||||
num_ctx: int,
|
||||
num_predict: int,
|
||||
) -> dict[str, Any]:
|
||||
case_dir = output_root / case["case_id"]
|
||||
case_dir.mkdir(parents=True, exist_ok=False)
|
||||
input_payload = {
|
||||
"case_id": case["case_id"],
|
||||
"description": case["description"],
|
||||
"evidence_units": case["evidence_units"],
|
||||
}
|
||||
(case_dir / "input.json").write_text(
|
||||
json.dumps(input_payload, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
prompt = build_prompt(case)
|
||||
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
|
||||
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
raw_text, metadata = call_ollama(
|
||||
endpoint, model, prompt, timeout, num_ctx, num_predict
|
||||
)
|
||||
(case_dir / "raw_model_response.txt").write_text(
|
||||
raw_text + "\n", encoding="utf-8"
|
||||
)
|
||||
(case_dir / "ollama_metadata.json").write_text(
|
||||
json.dumps(metadata, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
parsed = parse_model_json(raw_text)
|
||||
(case_dir / "parsed_output.json").write_text(
|
||||
json.dumps(parsed, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
validated = validate_reconstruction(parsed, case["evidence_units"])
|
||||
evaluation = evaluate_reconstruction(validated, case["expected"])
|
||||
except requests.RequestException as exc:
|
||||
failure = {
|
||||
"case_id": case["case_id"],
|
||||
"error_type": type(exc).__name__,
|
||||
"error": str(exc),
|
||||
"elapsed_seconds": round(time.perf_counter() - started, 3),
|
||||
}
|
||||
(case_dir / "validation_failure.json").write_text(
|
||||
json.dumps(failure, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
raise
|
||||
except (json.JSONDecodeError, ReconstructionValidationError, ValueError) as exc:
|
||||
elapsed = round(time.perf_counter() - started, 3)
|
||||
failure = {
|
||||
"case_id": case["case_id"],
|
||||
"error_type": type(exc).__name__,
|
||||
"error": str(exc),
|
||||
"elapsed_seconds": elapsed,
|
||||
}
|
||||
(case_dir / "validation_failure.json").write_text(
|
||||
json.dumps(failure, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
result = {
|
||||
"case_id": case["case_id"],
|
||||
"description": case["description"],
|
||||
"verdict": "FAIL",
|
||||
"reason": f"{type(exc).__name__}: {exc}",
|
||||
"passed_checks": 0,
|
||||
"check_count": 0,
|
||||
"critical_failures": ["schema_validation"],
|
||||
"checks": [],
|
||||
"elapsed_seconds": elapsed,
|
||||
"subject_titles": [],
|
||||
}
|
||||
(case_dir / "evaluation.json").write_text(
|
||||
json.dumps(result, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
return result
|
||||
|
||||
result = {
|
||||
"case_id": case["case_id"],
|
||||
"description": case["description"],
|
||||
**evaluation,
|
||||
"elapsed_seconds": metadata["elapsed_seconds"],
|
||||
"subject_titles": [item["title"] for item in validated["subjects"]],
|
||||
}
|
||||
(case_dir / "evaluation.json").write_text(
|
||||
json.dumps(result, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
|
||||
cases = load_fixture(args.fixture)
|
||||
selected = set(args.case_ids or [])
|
||||
if selected:
|
||||
known = {case["case_id"] for case in cases}
|
||||
unknown = selected - known
|
||||
if unknown:
|
||||
raise ValueError(f"unknown requested case IDs: {sorted(unknown)}")
|
||||
cases = [case for case in cases if case["case_id"] in selected]
|
||||
|
||||
args.output.mkdir(parents=True, exist_ok=False)
|
||||
results: list[dict[str, Any]] = []
|
||||
started = time.perf_counter()
|
||||
for index, case in enumerate(cases, start=1):
|
||||
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
|
||||
results.append(
|
||||
run_case(
|
||||
case,
|
||||
args.output,
|
||||
args.endpoint,
|
||||
args.model,
|
||||
args.timeout,
|
||||
args.num_ctx,
|
||||
args.num_predict,
|
||||
)
|
||||
)
|
||||
summary = {
|
||||
"experiment": "topic_reconstruction_v2",
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"model": args.model,
|
||||
"temperature": 0,
|
||||
"think": False,
|
||||
"case_count": len(cases),
|
||||
"llm_call_count": len(results),
|
||||
"runtime_seconds": round(time.perf_counter() - started, 3),
|
||||
"verdict_counts": {
|
||||
verdict: sum(item["verdict"] == verdict for item in results)
|
||||
for verdict in ("PASS", "PARTIAL", "FAIL")
|
||||
},
|
||||
"results": results,
|
||||
}
|
||||
(args.output / "summary.json").write_text(
|
||||
json.dumps(summary, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
return summary
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
try:
|
||||
summary = run_experiment(args)
|
||||
except (OSError, ValueError, requests.RequestException) as exc:
|
||||
print(f"Error: {exc}")
|
||||
return 1
|
||||
print(json.dumps(summary["verdict_counts"], sort_keys=True))
|
||||
print(f"Artifacts: {args.output.resolve()}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user