Add evidence-near semantic architecture experiments
Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
This commit is contained in:
@@ -0,0 +1,344 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Extract reduced-semantic-load evidence-near observations."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
SCHEMA_VERSION = "experimental-evidence-observations-v2"
|
||||
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
|
||||
DEFAULT_MODEL = "qwen3.5:9B"
|
||||
MODALITIES = {"factual", "possible", "suggested", "interpersonal_request", "impersonal_necessity", "information_question", "committed"}
|
||||
TEMPORALITIES = {"existing", "future", "completed", "unspecified"}
|
||||
EVALUATIONS = {"positive", "negative", "none"}
|
||||
BINARY_SIGNALS = {"explicit", "absent"}
|
||||
PRESENCE_SIGNALS = {"present", "absent"}
|
||||
CLARIFICATION_NEEDS = {"explicit", "implicit", "none"}
|
||||
OBSERVATION_ID_RE = re.compile(r"^obs_[1-9][0-9]*$")
|
||||
|
||||
|
||||
class ObservationValidationError(ValueError):
|
||||
"""Raised for invalid fixtures or model output."""
|
||||
|
||||
|
||||
PROMPT_TEMPLATE = """You extract atomic linguistic and discourse observations for one fixed Discussion Subject.
|
||||
|
||||
Preserve only facts directly expressed by the evidence. Do not derive responsibility,
|
||||
agreement, decisions, action items, open questions, accepted trials, rejected
|
||||
alternatives, established actions, or protocol eligibility. Speaker identity, a name,
|
||||
an addressee, first-person language, collective "we", and impersonal "man" never by
|
||||
themselves establish responsibility.
|
||||
|
||||
Return exactly one JSON object with this shape:
|
||||
{{
|
||||
"schema_version": "experimental-evidence-observations-v2",
|
||||
"subject_id": "copy exactly",
|
||||
"subject": "copy exactly",
|
||||
"observations": [
|
||||
{{
|
||||
"observation_id": "obs_1",
|
||||
"evidence_id": "e1",
|
||||
"content": "directly supported atomic observation",
|
||||
"refers_to": null,
|
||||
"speaker": "name copied from evidence or null",
|
||||
"named_person": null,
|
||||
"addressee": null,
|
||||
"self_reference": false,
|
||||
"collective_we": false,
|
||||
"impersonal_person_reference": false,
|
||||
"modality": "factual",
|
||||
"temporality": "existing",
|
||||
"evaluation": "none",
|
||||
"affirmation": "absent",
|
||||
"negation": "absent",
|
||||
"determination_statement": "absent",
|
||||
"uncertainty": "absent",
|
||||
"clarification_need": "none",
|
||||
"qualifier": null,
|
||||
"limits_target": null
|
||||
}}
|
||||
]
|
||||
}}
|
||||
|
||||
Rules:
|
||||
- Produce multiple observations for distinct propositions in one evidence unit, but do
|
||||
not fragment a single proposition unnecessarily.
|
||||
- observation_id is sequential in evidence order. evidence_id must be copied exactly.
|
||||
- refers_to is null or one earlier observation_id when the utterance explicitly refers
|
||||
to it. Never use arrays. Preserve joint-reference utterances without inventing a
|
||||
multi-target graph.
|
||||
- speaker is the explicit transcript speaker. named_person is a person explicitly
|
||||
named in the proposition. addressee is a person explicitly addressed.
|
||||
- self_reference marks singular first-person self-reference. collective_we marks
|
||||
collective first-person language. impersonal_person_reference marks impersonal
|
||||
person expressions such as German "man".
|
||||
- modality is factual, possible, suggested, interpersonal_request,
|
||||
impersonal_necessity, information_question, or committed.
|
||||
- temporality is existing, future, completed, or unspecified.
|
||||
- evaluation is positive, negative, or none, only when linguistically supported.
|
||||
- affirmation is explicit only for an explicit affirmative discourse signal such as
|
||||
"ja". negation is explicit only for directly expressed negation/rejection.
|
||||
- determination_statement is present only when the utterance explicitly says a
|
||||
determination has been made.
|
||||
- uncertainty is present or absent. clarification_need is explicit, implicit, or none.
|
||||
- qualifier is null or concise evidence-grounded qualifying text.
|
||||
- limits_target is null or one earlier observation explicitly limited in validity or
|
||||
scope by this observation.
|
||||
- Use JSON null, never the string "null". Output no fields beyond the schema.
|
||||
|
||||
Fixed Gold input:
|
||||
{input_json}
|
||||
"""
|
||||
|
||||
|
||||
OBSERVATION_KEYS = {
|
||||
"observation_id", "evidence_id", "content", "refers_to", "speaker",
|
||||
"named_person", "addressee", "self_reference", "collective_we",
|
||||
"impersonal_person_reference", "modality", "temporality", "evaluation",
|
||||
"affirmation", "negation", "determination_statement", "uncertainty",
|
||||
"clarification_need", "qualifier", "limits_target",
|
||||
}
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("fixture", type=Path)
|
||||
parser.add_argument("-o", "--output", type=Path, required=True)
|
||||
parser.add_argument("--model", default=DEFAULT_MODEL)
|
||||
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
|
||||
parser.add_argument("--timeout", type=int, default=300)
|
||||
parser.add_argument("--num-ctx", type=int, default=16384)
|
||||
parser.add_argument("--num-predict", type=int, default=4096)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None:
|
||||
missing, unknown = required - value.keys(), value.keys() - required
|
||||
if missing:
|
||||
raise ObservationValidationError(f"{location} missing required keys: {sorted(missing)}")
|
||||
if unknown:
|
||||
raise ObservationValidationError(f"{location} has unknown keys: {sorted(unknown)}")
|
||||
|
||||
|
||||
def _text(value: Any, location: str) -> str:
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise ObservationValidationError(f"{location} must be a non-empty string")
|
||||
result = value.strip()
|
||||
if result.casefold() == "null":
|
||||
raise ObservationValidationError(f"{location} must not be the string 'null'")
|
||||
return result
|
||||
|
||||
|
||||
def _nullable_text(value: Any, location: str) -> None:
|
||||
if value is not None:
|
||||
_text(value, location)
|
||||
|
||||
|
||||
def _prior_reference(value: Any, location: str, earlier: set[str]) -> None:
|
||||
if value is None:
|
||||
return
|
||||
reference = _text(value, location)
|
||||
if reference not in earlier:
|
||||
raise ObservationValidationError(f"{location} references unknown or later observation: {reference}")
|
||||
|
||||
|
||||
def validate_observations(data: Any, case: dict[str, Any]) -> dict[str, Any]:
|
||||
validate_case(case)
|
||||
if not isinstance(data, dict):
|
||||
raise ObservationValidationError("output must be an object")
|
||||
_exact_keys(data, {"schema_version", "subject_id", "subject", "observations"}, "output")
|
||||
if data["schema_version"] != SCHEMA_VERSION:
|
||||
raise ObservationValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
|
||||
if data["subject_id"] != case["subject_id"] or data["subject"] != case["subject"]:
|
||||
raise ObservationValidationError("model changed the fixed Discussion Subject")
|
||||
observations = data["observations"]
|
||||
if not isinstance(observations, list) or not observations:
|
||||
raise ObservationValidationError("output.observations must be a non-empty array")
|
||||
known_evidence = {item["evidence_id"] for item in case["evidence"]}
|
||||
earlier: set[str] = set()
|
||||
for index, observation in enumerate(observations, 1):
|
||||
location = f"output.observations[{index - 1}]"
|
||||
if not isinstance(observation, dict):
|
||||
raise ObservationValidationError(f"{location} must be an object")
|
||||
_exact_keys(observation, OBSERVATION_KEYS, location)
|
||||
observation_id = _text(observation["observation_id"], f"{location}.observation_id")
|
||||
if not OBSERVATION_ID_RE.fullmatch(observation_id) or observation_id != f"obs_{index}":
|
||||
raise ObservationValidationError(f"{location}.observation_id must be obs_{index}")
|
||||
evidence_id = _text(observation["evidence_id"], f"{location}.evidence_id")
|
||||
if evidence_id not in known_evidence:
|
||||
raise ObservationValidationError(f"{location}.evidence_id references unknown evidence: {evidence_id}")
|
||||
_text(observation["content"], f"{location}.content")
|
||||
_prior_reference(observation["refers_to"], f"{location}.refers_to", earlier)
|
||||
_prior_reference(observation["limits_target"], f"{location}.limits_target", earlier)
|
||||
for field in ("speaker", "named_person", "addressee", "qualifier"):
|
||||
_nullable_text(observation[field], f"{location}.{field}")
|
||||
for field in ("self_reference", "collective_we", "impersonal_person_reference"):
|
||||
if not isinstance(observation[field], bool):
|
||||
raise ObservationValidationError(f"{location}.{field} must be boolean")
|
||||
for field, values in (
|
||||
("modality", MODALITIES), ("temporality", TEMPORALITIES),
|
||||
("evaluation", EVALUATIONS), ("affirmation", BINARY_SIGNALS),
|
||||
("negation", BINARY_SIGNALS), ("determination_statement", PRESENCE_SIGNALS),
|
||||
("uncertainty", PRESENCE_SIGNALS), ("clarification_need", CLARIFICATION_NEEDS),
|
||||
):
|
||||
if observation[field] not in values:
|
||||
raise ObservationValidationError(f"{location}.{field} is invalid: {observation[field]!r}")
|
||||
earlier.add(observation_id)
|
||||
return data
|
||||
|
||||
|
||||
def validate_case(case: Any) -> dict[str, Any]:
|
||||
if not isinstance(case, dict):
|
||||
raise ObservationValidationError("case must be an object")
|
||||
_exact_keys(case, {"case_id", "description", "subject_id", "subject", "evidence", "expected_observations"}, "case")
|
||||
for field in ("case_id", "description", "subject_id", "subject"):
|
||||
_text(case[field], f"case.{field}")
|
||||
if not isinstance(case["evidence"], list) or not case["evidence"]:
|
||||
raise ObservationValidationError("case.evidence must be a non-empty array")
|
||||
seen: set[str] = set()
|
||||
for index, unit in enumerate(case["evidence"]):
|
||||
_exact_keys(unit, {"evidence_id", "text"}, f"case.evidence[{index}]")
|
||||
evidence_id = _text(unit["evidence_id"], f"case.evidence[{index}].evidence_id")
|
||||
if evidence_id in seen:
|
||||
raise ObservationValidationError(f"duplicate evidence ID: {evidence_id}")
|
||||
seen.add(evidence_id)
|
||||
_text(unit["text"], f"case.evidence[{index}].text")
|
||||
if not isinstance(case["expected_observations"], list) or not case["expected_observations"]:
|
||||
raise ObservationValidationError("case.expected_observations must be a non-empty array")
|
||||
return case
|
||||
|
||||
|
||||
def validate_fixture_case(case: dict[str, Any]) -> dict[str, Any]:
|
||||
validate_case(case)
|
||||
validate_observations({"schema_version": SCHEMA_VERSION, "subject_id": case["subject_id"], "subject": case["subject"], "observations": case["expected_observations"]}, case)
|
||||
return case
|
||||
|
||||
|
||||
def build_prompt(case: dict[str, Any]) -> str:
|
||||
validate_fixture_case(case)
|
||||
model_input = {key: case[key] for key in ("subject_id", "subject", "evidence")}
|
||||
return PROMPT_TEMPLATE.format(input_json=json.dumps(model_input, ensure_ascii=False, indent=2))
|
||||
|
||||
|
||||
def parse_model_json(raw_text: str) -> dict[str, Any]:
|
||||
data = json.loads(raw_text)
|
||||
if not isinstance(data, dict):
|
||||
raise ObservationValidationError("model response JSON must be an object")
|
||||
return data
|
||||
|
||||
|
||||
def build_ollama_payload(model: str, prompt: str, num_ctx: int, num_predict: int) -> dict[str, Any]:
|
||||
return {"model": model, "prompt": prompt, "think": False, "stream": False, "format": "json", "options": {"temperature": 0, "num_ctx": num_ctx, "num_predict": num_predict}}
|
||||
|
||||
|
||||
def call_ollama(endpoint: str, model: str, prompt: str, timeout: int, num_ctx: int, num_predict: int) -> tuple[str, dict[str, Any]]:
|
||||
started = time.perf_counter()
|
||||
response = requests.post(endpoint, json=build_ollama_payload(model, prompt, num_ctx, num_predict), timeout=timeout)
|
||||
elapsed = time.perf_counter() - started
|
||||
response.raise_for_status()
|
||||
body = response.json()
|
||||
raw = body.get("response") if isinstance(body, dict) else None
|
||||
if not isinstance(raw, str) or not raw.strip():
|
||||
raise ValueError("Ollama returned no usable response text")
|
||||
metadata = {"model": body.get("model", model), "elapsed_seconds": round(elapsed, 3), "total_duration_ns": body.get("total_duration"), "load_duration_ns": body.get("load_duration"), "prompt_eval_count": body.get("prompt_eval_count"), "prompt_eval_duration_ns": body.get("prompt_eval_duration"), "eval_count": body.get("eval_count"), "eval_duration_ns": body.get("eval_duration"), "configuration": {"temperature": 0, "think": False, "num_ctx": num_ctx, "num_predict": num_predict, "retries": 0}}
|
||||
return raw.strip(), metadata
|
||||
|
||||
|
||||
COMPARE_FIELDS = tuple(sorted(OBSERVATION_KEYS - {"observation_id", "content", "qualifier"}))
|
||||
|
||||
|
||||
def _qualifier_matches(actual: str | None, expected: str | None) -> bool:
|
||||
if expected is None:
|
||||
return actual is None
|
||||
if actual is None:
|
||||
return False
|
||||
return any(term.strip().casefold() in actual.casefold() for term in expected.split("|"))
|
||||
|
||||
|
||||
def evaluate_observations(data: dict[str, Any], expected: list[dict[str, Any]]) -> dict[str, Any]:
|
||||
actual = data["observations"]
|
||||
checks = [{"name": "observation_count", "passed": len(actual) == len(expected), "critical": False}]
|
||||
for index, (got, want) in enumerate(zip(actual, expected), 1):
|
||||
for field in COMPARE_FIELDS:
|
||||
checks.append({"name": f"obs_{index}:{field}", "passed": got[field] == want[field], "critical": field in {"evidence_id", "refers_to", "limits_target", "modality", "affirmation", "negation", "determination_statement"}})
|
||||
checks.append({"name": f"obs_{index}:qualifier", "passed": _qualifier_matches(got["qualifier"], want["qualifier"]), "critical": False})
|
||||
passed = sum(check["passed"] for check in checks)
|
||||
ratio = passed / len(checks)
|
||||
critical = [check["name"] for check in checks if check["critical"] and not check["passed"]]
|
||||
verdict = "PASS" if ratio == 1 else "PARTIAL" if ratio >= 0.75 and not critical else "FAIL"
|
||||
return {"verdict": verdict, "matched_checks": passed, "check_count": len(checks), "match_ratio": round(ratio, 3), "critical_failures": critical, "checks": checks}
|
||||
|
||||
|
||||
def load_fixture(path: Path) -> list[dict[str, Any]]:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
if not isinstance(data, dict) or set(data) != {"cases"} or not isinstance(data["cases"], list) or not data["cases"]:
|
||||
raise ObservationValidationError("fixture must contain exactly one non-empty cases list")
|
||||
seen: set[str] = set()
|
||||
for case in data["cases"]:
|
||||
validate_fixture_case(case)
|
||||
if case["case_id"] in seen:
|
||||
raise ObservationValidationError(f"duplicate case ID: {case['case_id']}")
|
||||
seen.add(case["case_id"])
|
||||
return data["cases"]
|
||||
|
||||
|
||||
def _write_json(path: Path, value: Any) -> None:
|
||||
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def run_case(case: dict[str, Any], output_root: Path, endpoint: str, model: str, timeout: int, num_ctx: int, num_predict: int) -> dict[str, Any]:
|
||||
case_dir = output_root / case["case_id"]
|
||||
case_dir.mkdir(parents=True, exist_ok=False)
|
||||
_write_json(case_dir / "gold_input.json", {key: case[key] for key in ("case_id", "description", "subject_id", "subject", "evidence")})
|
||||
_write_json(case_dir / "gold_expected_observations.json", case["expected_observations"])
|
||||
prompt = build_prompt(case)
|
||||
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
|
||||
started = time.perf_counter()
|
||||
raw, metadata = call_ollama(endpoint, model, prompt, timeout, num_ctx, num_predict)
|
||||
(case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8")
|
||||
_write_json(case_dir / "ollama_metadata.json", metadata)
|
||||
try:
|
||||
parsed = parse_model_json(raw)
|
||||
_write_json(case_dir / "parsed_observations.json", parsed)
|
||||
validate_observations(parsed, case)
|
||||
validation = {"valid": True, "error": None}
|
||||
evaluation = evaluate_observations(parsed, case["expected_observations"])
|
||||
except (json.JSONDecodeError, ObservationValidationError, ValueError) as exc:
|
||||
validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)}
|
||||
evaluation = {"verdict": "FAIL", "matched_checks": 0, "check_count": 0, "match_ratio": 0, "critical_failures": ["schema_validation"], "checks": []}
|
||||
_write_json(case_dir / "validation_result.json", validation)
|
||||
result = {"case_id": case["case_id"], **evaluation, "elapsed_seconds": round(time.perf_counter() - started, 3)}
|
||||
_write_json(case_dir / "evaluation_result.json", result)
|
||||
return result
|
||||
|
||||
|
||||
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
|
||||
cases = load_fixture(args.fixture)
|
||||
args.output.mkdir(parents=True, exist_ok=False)
|
||||
started = time.perf_counter()
|
||||
results = []
|
||||
for index, case in enumerate(cases, 1):
|
||||
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
|
||||
results.append(run_case(case, args.output, args.endpoint, args.model, args.timeout, args.num_ctx, args.num_predict))
|
||||
summary = {"experiment": "evidence_near_observation_extraction_v2", "schema_version": SCHEMA_VERSION, "model": args.model, "temperature": 0, "think": False, "retries": 0, "case_count": len(cases), "llm_call_count": len(results), "runtime_seconds": round(time.perf_counter() - started, 3), "verdict_counts": {verdict: sum(result["verdict"] == verdict for result in results) for verdict in ("PASS", "PARTIAL", "FAIL")}, "results": results}
|
||||
_write_json(args.output / "summary.json", summary)
|
||||
return summary
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
summary = run_experiment(args)
|
||||
print(json.dumps(summary, ensure_ascii=False, indent=2))
|
||||
return 0 if summary["verdict_counts"]["FAIL"] == 0 else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user