Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
396 lines
20 KiB
Python
396 lines
20 KiB
Python
#!/usr/bin/env python3
|
|
"""Extract evidence-near observations for a fixed Discussion Subject."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import re
|
|
import time
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
import requests
|
|
|
|
|
|
SCHEMA_VERSION = "experimental-evidence-observations-v1"
|
|
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
|
|
DEFAULT_MODEL = "qwen3.5:9B"
|
|
DEFAULT_TIMEOUT = 300
|
|
DEFAULT_NUM_CTX = 16384
|
|
DEFAULT_NUM_PREDICT = 4096
|
|
|
|
RELATIONS = {"none", "supports", "opposes", "qualifies", "limits_scope"}
|
|
MODALITIES = {
|
|
"factual",
|
|
"possible",
|
|
"suggested",
|
|
"interpersonal_request",
|
|
"impersonal_necessity",
|
|
"information_question",
|
|
"committed",
|
|
}
|
|
TEMPORALITIES = {"existing", "future", "completed", "unspecified"}
|
|
EVALUATIONS = {"positive", "negative", "none"}
|
|
AGREEMENTS = {"accepted", "rejected", "unclear", "none"}
|
|
RESPONSIBILITIES = {"none", "named", "accepted"}
|
|
UNCERTAINTIES = {"present", "absent"}
|
|
CLARIFICATION_NEEDS = {"explicit", "implicit", "none"}
|
|
OBSERVATION_ID_RE = re.compile(r"^obs_[1-9][0-9]*$")
|
|
|
|
|
|
class ObservationValidationError(ValueError):
|
|
"""Raised when an experimental fixture or model output is invalid."""
|
|
|
|
|
|
PROMPT_TEMPLATE = """You extract atomic, evidence-near observations for one fixed Discussion Subject.
|
|
|
|
Stop before protocol interpretation. Never classify anything as an idea, proposal,
|
|
objection, decision, action item, or open question. Do not determine protocol
|
|
eligibility, reconstruct topics, generate a protocol, or invent missing stages.
|
|
|
|
Split an evidence unit into multiple observations when it directly contains multiple
|
|
propositions. Preserve every observation's source evidence ID. Use concise content in
|
|
the evidence language.
|
|
|
|
Return exactly one JSON object with this shape:
|
|
{{
|
|
"schema_version": "experimental-evidence-observations-v1",
|
|
"subject_id": "copy exactly",
|
|
"subject": "copy exactly",
|
|
"observations": [
|
|
{{
|
|
"observation_id": "obs_1",
|
|
"evidence_id": "e1",
|
|
"content": "directly supported atomic observation",
|
|
"target": "discussion_subject",
|
|
"relation": "none",
|
|
"modality": "factual",
|
|
"temporality": "existing",
|
|
"evaluation": "none",
|
|
"agreement": "none",
|
|
"responsibility": "none",
|
|
"person": null,
|
|
"uncertainty": "absent",
|
|
"clarification_need": "none",
|
|
"scope": "absent"
|
|
}}
|
|
]
|
|
}}
|
|
|
|
Rules:
|
|
- Number observation_id sequentially as obs_1, obs_2, ... in evidence order.
|
|
- target is "discussion_subject", one earlier observation_id, or a non-empty list of
|
|
earlier observation_ids only when the evidence jointly refers to them.
|
|
- relation is only none, supports, opposes, qualifies, or limits_scope.
|
|
- modality is only factual, possible, suggested, interpersonal_request,
|
|
impersonal_necessity, information_question, or committed.
|
|
- interpersonal_request is a direct request to another person.
|
|
- impersonal_necessity says something needs to happen without assigning it.
|
|
- information_question expresses missing information without assigning work.
|
|
- temporality is only existing, future, completed, or unspecified.
|
|
- evaluation is positive, negative, or none. Do not infer evaluation from world
|
|
knowledge. A bare cost or technical fact normally has evaluation none.
|
|
- agreement is only accepted, rejected, unclear, or none and applies to target.
|
|
- responsibility is none, named, or accepted. Use named only for an explicitly
|
|
addressed candidate and accepted only for explicit acceptance/commitment.
|
|
- person is the explicit person's name for named/accepted responsibility; otherwise
|
|
use JSON null. Mentioning or speaking in first person does not establish ownership.
|
|
- uncertainty is present or absent.
|
|
- clarification_need is explicit, implicit, or none.
|
|
- scope is an evidence-grounded qualifier, or exactly "absent". Never use null or the
|
|
string "null" anywhere.
|
|
- Confirmation of a rejection targets the rejection observation, not the option.
|
|
- A trial-only qualification targets and limits the accepted trial.
|
|
- A negative consequence can oppose another observation without requiring
|
|
clarification.
|
|
- Personal preference is not group rejection.
|
|
- Collective "we" does not name an individual owner.
|
|
|
|
Fixed Gold input:
|
|
{input_json}
|
|
"""
|
|
|
|
|
|
def parse_args() -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(description="Run the evidence-observation Gold experiment.")
|
|
parser.add_argument("fixture", type=Path)
|
|
parser.add_argument("-o", "--output", type=Path, required=True)
|
|
parser.add_argument("--model", default=DEFAULT_MODEL)
|
|
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
|
|
parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT)
|
|
parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX)
|
|
parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT)
|
|
return parser.parse_args()
|
|
|
|
|
|
def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None:
|
|
missing = required - value.keys()
|
|
unknown = value.keys() - required
|
|
if missing:
|
|
raise ObservationValidationError(f"{location} missing required keys: {sorted(missing)}")
|
|
if unknown:
|
|
raise ObservationValidationError(f"{location} has unknown keys: {sorted(unknown)}")
|
|
|
|
|
|
def _text(value: Any, location: str) -> str:
|
|
if not isinstance(value, str) or not value.strip():
|
|
raise ObservationValidationError(f"{location} must be a non-empty string")
|
|
result = value.strip()
|
|
if result.casefold() == "null":
|
|
raise ObservationValidationError(f"{location} must not be the string 'null'")
|
|
return result
|
|
|
|
|
|
OBSERVATION_KEYS = {
|
|
"observation_id", "evidence_id", "content", "target", "relation", "modality",
|
|
"temporality", "evaluation", "agreement", "responsibility", "person",
|
|
"uncertainty", "clarification_need", "scope",
|
|
}
|
|
|
|
|
|
def _validate_target(value: Any, location: str, earlier: set[str]) -> None:
|
|
if isinstance(value, str):
|
|
target = _text(value, location)
|
|
if target != "discussion_subject" and target not in earlier:
|
|
raise ObservationValidationError(f"{location} references unknown or later observation: {target}")
|
|
return
|
|
if not isinstance(value, list) or not value:
|
|
raise ObservationValidationError(f"{location} must be discussion_subject, an earlier observation ID, or a non-empty list")
|
|
if len(value) < 2:
|
|
raise ObservationValidationError(f"{location} list must contain at least two jointly referenced observations")
|
|
seen: set[str] = set()
|
|
for index, item in enumerate(value):
|
|
target = _text(item, f"{location}[{index}]")
|
|
if target not in earlier:
|
|
raise ObservationValidationError(f"{location}[{index}] references unknown or later observation: {target}")
|
|
if target in seen:
|
|
raise ObservationValidationError(f"{location} contains duplicate target: {target}")
|
|
seen.add(target)
|
|
|
|
|
|
def validate_observations(data: Any, case: dict[str, Any]) -> dict[str, Any]:
|
|
validate_case(case)
|
|
if not isinstance(data, dict):
|
|
raise ObservationValidationError("output must be an object")
|
|
_exact_keys(data, {"schema_version", "subject_id", "subject", "observations"}, "output")
|
|
if data["schema_version"] != SCHEMA_VERSION:
|
|
raise ObservationValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
|
|
if data["subject_id"] != case["subject_id"] or data["subject"] != case["subject"]:
|
|
raise ObservationValidationError("model changed the fixed Discussion Subject")
|
|
observations = data["observations"]
|
|
if not isinstance(observations, list) or not observations:
|
|
raise ObservationValidationError("output.observations must be a non-empty array")
|
|
known_evidence = {item["evidence_id"] for item in case["evidence"]}
|
|
earlier: set[str] = set()
|
|
for index, observation in enumerate(observations, start=1):
|
|
location = f"output.observations[{index - 1}]"
|
|
if not isinstance(observation, dict):
|
|
raise ObservationValidationError(f"{location} must be an object")
|
|
_exact_keys(observation, OBSERVATION_KEYS, location)
|
|
observation_id = _text(observation["observation_id"], f"{location}.observation_id")
|
|
if not OBSERVATION_ID_RE.fullmatch(observation_id) or observation_id != f"obs_{index}":
|
|
raise ObservationValidationError(f"{location}.observation_id must be obs_{index}")
|
|
evidence_id = _text(observation["evidence_id"], f"{location}.evidence_id")
|
|
if evidence_id not in known_evidence:
|
|
raise ObservationValidationError(f"{location}.evidence_id references unknown evidence: {evidence_id}")
|
|
_text(observation["content"], f"{location}.content")
|
|
_validate_target(observation["target"], f"{location}.target", earlier)
|
|
for field, values in (
|
|
("relation", RELATIONS), ("modality", MODALITIES),
|
|
("temporality", TEMPORALITIES), ("evaluation", EVALUATIONS),
|
|
("agreement", AGREEMENTS), ("responsibility", RESPONSIBILITIES),
|
|
("uncertainty", UNCERTAINTIES), ("clarification_need", CLARIFICATION_NEEDS),
|
|
):
|
|
if observation[field] not in values:
|
|
raise ObservationValidationError(f"{location}.{field} is invalid: {observation[field]!r}")
|
|
person = observation["person"]
|
|
if observation["responsibility"] == "none":
|
|
if person is not None:
|
|
raise ObservationValidationError(f"{location}.person must be JSON null when responsibility is none")
|
|
else:
|
|
_text(person, f"{location}.person")
|
|
scope = _text(observation["scope"], f"{location}.scope")
|
|
if scope.casefold() == "null":
|
|
raise ObservationValidationError(f"{location}.scope must use 'absent', not 'null'")
|
|
earlier.add(observation_id)
|
|
return data
|
|
|
|
|
|
def validate_case(case: Any) -> dict[str, Any]:
|
|
if not isinstance(case, dict):
|
|
raise ObservationValidationError("case must be an object")
|
|
_exact_keys(case, {"case_id", "description", "subject_id", "subject", "evidence", "expected_observations"}, "case")
|
|
for field in ("case_id", "description", "subject_id", "subject"):
|
|
_text(case[field], f"case.{field}")
|
|
evidence = case["evidence"]
|
|
if not isinstance(evidence, list) or not evidence:
|
|
raise ObservationValidationError("case.evidence must be a non-empty array")
|
|
seen: set[str] = set()
|
|
for index, unit in enumerate(evidence):
|
|
location = f"case.evidence[{index}]"
|
|
if not isinstance(unit, dict):
|
|
raise ObservationValidationError(f"{location} must be an object")
|
|
_exact_keys(unit, {"evidence_id", "text"}, location)
|
|
evidence_id = _text(unit["evidence_id"], f"{location}.evidence_id")
|
|
if evidence_id in seen:
|
|
raise ObservationValidationError(f"duplicate evidence ID: {evidence_id}")
|
|
seen.add(evidence_id)
|
|
_text(unit["text"], f"{location}.text")
|
|
expected = case["expected_observations"]
|
|
if not isinstance(expected, list) or not expected:
|
|
raise ObservationValidationError("case.expected_observations must be a non-empty array")
|
|
return case
|
|
|
|
|
|
def validate_fixture_case(case: dict[str, Any]) -> dict[str, Any]:
|
|
validate_case(case)
|
|
data = {"schema_version": SCHEMA_VERSION, "subject_id": case["subject_id"], "subject": case["subject"], "observations": case["expected_observations"]}
|
|
validate_observations(data, case)
|
|
return case
|
|
|
|
|
|
def build_prompt(case: dict[str, Any]) -> str:
|
|
validate_fixture_case(case)
|
|
model_input = {"subject_id": case["subject_id"], "subject": case["subject"], "evidence": case["evidence"]}
|
|
return PROMPT_TEMPLATE.format(input_json=json.dumps(model_input, ensure_ascii=False, indent=2))
|
|
|
|
|
|
def parse_model_json(raw_text: str) -> dict[str, Any]:
|
|
data = json.loads(raw_text)
|
|
if not isinstance(data, dict):
|
|
raise ObservationValidationError("model response JSON must be an object")
|
|
return data
|
|
|
|
|
|
def build_ollama_payload(model: str, prompt: str, num_ctx: int, num_predict: int) -> dict[str, Any]:
|
|
return {"model": model, "prompt": prompt, "think": False, "stream": False, "format": "json", "options": {"temperature": 0, "num_ctx": num_ctx, "num_predict": num_predict}}
|
|
|
|
|
|
def call_ollama(endpoint: str, model: str, prompt: str, timeout: int, num_ctx: int, num_predict: int) -> tuple[str, dict[str, Any]]:
|
|
started = time.perf_counter()
|
|
response = requests.post(endpoint, json=build_ollama_payload(model, prompt, num_ctx, num_predict), timeout=timeout)
|
|
elapsed = time.perf_counter() - started
|
|
response.raise_for_status()
|
|
body = response.json()
|
|
raw_text = body.get("response") if isinstance(body, dict) else None
|
|
if not isinstance(raw_text, str) or not raw_text.strip():
|
|
raise ValueError("Ollama returned no usable response text")
|
|
metadata = {
|
|
"model": body.get("model", model), "elapsed_seconds": round(elapsed, 3),
|
|
"total_duration_ns": body.get("total_duration"), "load_duration_ns": body.get("load_duration"),
|
|
"prompt_eval_count": body.get("prompt_eval_count"), "prompt_eval_duration_ns": body.get("prompt_eval_duration"),
|
|
"eval_count": body.get("eval_count"), "eval_duration_ns": body.get("eval_duration"),
|
|
"configuration": {"temperature": 0, "think": False, "num_ctx": num_ctx, "num_predict": num_predict, "retries": 0},
|
|
}
|
|
return raw_text.strip(), metadata
|
|
|
|
|
|
COMPARE_FIELDS = ("evidence_id", "target", "relation", "modality", "temporality", "evaluation", "agreement", "responsibility", "person", "uncertainty", "clarification_need")
|
|
|
|
|
|
def _scope_matches(actual: str, expected: str) -> bool:
|
|
if expected == "absent":
|
|
return actual == "absent"
|
|
expected_terms = [term.strip().casefold() for term in expected.split("|")]
|
|
folded = actual.casefold()
|
|
return any(term in folded for term in expected_terms)
|
|
|
|
|
|
def evaluate_observations(data: dict[str, Any], expected: list[dict[str, Any]]) -> dict[str, Any]:
|
|
actual = data["observations"]
|
|
checks: list[dict[str, Any]] = []
|
|
pair_count = min(len(actual), len(expected))
|
|
checks.append({"name": "observation_count", "passed": len(actual) == len(expected), "critical": False})
|
|
categories = {"missing_observations": max(0, len(expected) - len(actual)), "invented_observations": max(0, len(actual) - len(expected)), "stronger_commitment": 0, "weaker_commitment": 0, "incorrect_targets_relations": 0, "incorrect_responsibility": 0, "incorrect_uncertainty_clarification": 0}
|
|
commitment_rank = {"factual": 0, "possible": 1, "suggested": 1, "information_question": 1, "impersonal_necessity": 2, "interpersonal_request": 2, "committed": 3}
|
|
for index in range(pair_count):
|
|
got, want = actual[index], expected[index]
|
|
for field in COMPARE_FIELDS:
|
|
passed = got[field] == want[field]
|
|
checks.append({"name": f"obs_{index + 1}:{field}", "passed": passed, "critical": field in {"evidence_id", "target", "relation", "modality", "agreement", "responsibility", "person"}})
|
|
if not passed:
|
|
if field in {"target", "relation"}: categories["incorrect_targets_relations"] += 1
|
|
if field in {"responsibility", "person"}: categories["incorrect_responsibility"] += 1
|
|
if field in {"uncertainty", "clarification_need"}: categories["incorrect_uncertainty_clarification"] += 1
|
|
scope_ok = _scope_matches(got["scope"], want["scope"])
|
|
checks.append({"name": f"obs_{index + 1}:scope", "passed": scope_ok, "critical": False})
|
|
got_rank, want_rank = commitment_rank[got["modality"]], commitment_rank[want["modality"]]
|
|
if got_rank > want_rank or (want["agreement"] == "none" and got["agreement"] in {"accepted", "rejected"}): categories["stronger_commitment"] += 1
|
|
if got_rank < want_rank or (want["agreement"] in {"accepted", "rejected"} and got["agreement"] == "none"): categories["weaker_commitment"] += 1
|
|
passed_count = sum(check["passed"] for check in checks)
|
|
critical_failures = [check["name"] for check in checks if check["critical"] and not check["passed"]]
|
|
ratio = passed_count / len(checks)
|
|
if ratio == 1:
|
|
verdict = "PASS"
|
|
elif ratio >= 0.7 and categories["stronger_commitment"] == 0 and categories["incorrect_responsibility"] == 0:
|
|
verdict = "PARTIAL"
|
|
else:
|
|
verdict = "FAIL"
|
|
return {"verdict": verdict, "matched_checks": passed_count, "check_count": len(checks), "match_ratio": round(ratio, 3), "critical_failures": critical_failures, "error_categories": categories, "checks": checks}
|
|
|
|
|
|
def load_fixture(path: Path) -> list[dict[str, Any]]:
|
|
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
|
if not isinstance(data, dict) or set(data) != {"cases"} or not isinstance(data["cases"], list) or not data["cases"]:
|
|
raise ObservationValidationError("fixture must contain exactly one non-empty cases list")
|
|
seen: set[str] = set()
|
|
for case in data["cases"]:
|
|
validate_fixture_case(case)
|
|
if case["case_id"] in seen:
|
|
raise ObservationValidationError(f"duplicate case ID: {case['case_id']}")
|
|
seen.add(case["case_id"])
|
|
return data["cases"]
|
|
|
|
|
|
def _write_json(path: Path, value: Any) -> None:
|
|
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
|
|
|
|
|
|
def run_case(case: dict[str, Any], output_root: Path, endpoint: str, model: str, timeout: int, num_ctx: int, num_predict: int) -> dict[str, Any]:
|
|
case_dir = output_root / case["case_id"]
|
|
case_dir.mkdir(parents=True, exist_ok=False)
|
|
_write_json(case_dir / "gold_input.json", {key: case[key] for key in ("case_id", "description", "subject_id", "subject", "evidence")})
|
|
_write_json(case_dir / "gold_expected_observations.json", case["expected_observations"])
|
|
prompt = build_prompt(case)
|
|
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
|
|
started = time.perf_counter()
|
|
try:
|
|
raw, metadata = call_ollama(endpoint, model, prompt, timeout, num_ctx, num_predict)
|
|
(case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8")
|
|
_write_json(case_dir / "ollama_metadata.json", metadata)
|
|
parsed = parse_model_json(raw)
|
|
_write_json(case_dir / "parsed_observations.json", parsed)
|
|
validate_observations(parsed, case)
|
|
validation = {"valid": True, "error": None}
|
|
evaluation = evaluate_observations(parsed, case["expected_observations"])
|
|
except requests.RequestException:
|
|
raise
|
|
except (json.JSONDecodeError, ObservationValidationError, ValueError) as exc:
|
|
validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)}
|
|
evaluation = {"verdict": "FAIL", "matched_checks": 0, "check_count": 0, "match_ratio": 0, "critical_failures": ["schema_validation"], "error_categories": {}, "checks": []}
|
|
_write_json(case_dir / "validation_result.json", validation)
|
|
result = {"case_id": case["case_id"], **evaluation, "elapsed_seconds": round(time.perf_counter() - started, 3)}
|
|
_write_json(case_dir / "evaluation_result.json", result)
|
|
return result
|
|
|
|
|
|
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
|
|
cases = load_fixture(args.fixture)
|
|
args.output.mkdir(parents=True, exist_ok=False)
|
|
started = time.perf_counter()
|
|
results = []
|
|
for index, case in enumerate(cases, start=1):
|
|
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
|
|
results.append(run_case(case, args.output, args.endpoint, args.model, args.timeout, args.num_ctx, args.num_predict))
|
|
summary = {"experiment": "evidence_near_observation_extraction", "schema_version": SCHEMA_VERSION, "model": args.model, "temperature": 0, "think": False, "retries": 0, "case_count": len(cases), "llm_call_count": len(results), "runtime_seconds": round(time.perf_counter() - started, 3), "verdict_counts": {v: sum(r["verdict"] == v for r in results) for v in ("PASS", "PARTIAL", "FAIL")}, "results": results}
|
|
_write_json(args.output / "summary.json", summary)
|
|
return summary
|
|
|
|
|
|
def main() -> int:
|
|
args = parse_args()
|
|
summary = run_experiment(args)
|
|
print(json.dumps(summary, ensure_ascii=False, indent=2))
|
|
return 0 if summary["verdict_counts"]["FAIL"] == 0 else 1
|