Add evidence-near semantic architecture experiments

Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
This commit is contained in:
2026-08-19 15:46:22 +02:00
parent bcb197a908
commit 18beb3385f
29 changed files with 4542 additions and 0 deletions
@@ -0,0 +1 @@
"""Minimal semantic-preservation observation experiment."""
@@ -0,0 +1,271 @@
#!/usr/bin/env python3
"""Preserve meeting meaning as minimal atomic natural-language observations."""
from __future__ import annotations
import argparse
import json
import re
import time
from pathlib import Path
from typing import Any
import requests
SCHEMA_VERSION = "experimental-evidence-observations-v3"
DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate"
DEFAULT_MODEL = "qwen3.5:9B"
OBSERVATION_ID_RE = re.compile(r"^obs_[1-9][0-9]*$")
OBSERVATION_KEYS = {"observation_id", "evidence_id", "content", "speaker", "named_person", "addressee"}
class ObservationValidationError(ValueError):
"""Raised for invalid fixtures or model output."""
PROMPT_TEMPLATE = """Preserve the meeting meaning in atomic natural-language observations.
This is semantic preservation, not classification or summarization. Return only facts
faithfully contributed by the evidence. Conservative wording is more important than
elegant prose. When in doubt, preserve the source wording closely.
Return exactly one JSON object:
{{
"schema_version": "experimental-evidence-observations-v3",
"subject_id": "copy exactly",
"subject": "copy exactly",
"observations": [
{{
"observation_id": "obs_1",
"evidence_id": "e1",
"content": "atomic, semantically faithful observation",
"speaker": "speaker copied from evidence",
"named_person": null,
"addressee": null
}}
]
}}
Rules:
- Use only the six observation fields shown. Do not output classifications, labels,
relations, scope fields, responsibility, agreement, decisions, actions, questions,
eligibility, or any other field.
- observation_id is sequential in evidence order. Copy evidence_id and speaker.
- named_person is null or a person explicitly named in that observation's evidence.
- addressee is null or a person explicitly addressed in that observation's evidence.
- A name, speaker, or addressee never implies responsibility, acceptance, ownership,
or assignment.
- content is not a summary. Preserve distinctions needed for later interpretation:
maybe/perhaps; can/could; should/must; personal, collective, or impersonal wording;
explicit requests, acceptances, and rejections; uncertainty and unresolved status;
conditions such as "if at all"; quantities; deadlines; trial/process/comparison
boundaries; "not yet"; and sequence such as "then".
- Never strengthen modality, weaken uncertainty, turn possibility into fact, turn a
preference into group rejection, turn a request into established work, turn "we"
into individual ownership, remove conditions/limits, generalize, or invent relations.
- Split one evidence unit only when it contributes propositions that may later require
different interpretations. Do not split merely because it has several clauses.
- Do not emit observation-ID relations. When evidence clearly makes an observation
depend on the immediately preceding proposition, state that dependency naturally in
content, without inventing an antecedent.
- Preserve content in the evidence language. Use JSON null, never the string "null".
Fixed input:
{input_json}
"""
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser()
parser.add_argument("fixture", type=Path)
parser.add_argument("-o", "--output", type=Path, required=True)
parser.add_argument("--model", default=DEFAULT_MODEL)
parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT)
parser.add_argument("--timeout", type=int, default=300)
parser.add_argument("--num-ctx", type=int, default=16384)
parser.add_argument("--num-predict", type=int, default=4096)
return parser.parse_args()
def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None:
missing, unknown = required - value.keys(), value.keys() - required
if missing:
raise ObservationValidationError(f"{location} missing required keys: {sorted(missing)}")
if unknown:
raise ObservationValidationError(f"{location} has unknown keys: {sorted(unknown)}")
def _text(value: Any, location: str) -> str:
if not isinstance(value, str) or not value.strip():
raise ObservationValidationError(f"{location} must be a non-empty string")
result = value.strip()
if result.casefold() == "null":
raise ObservationValidationError(f"{location} must not be the string 'null'")
return result
def _explicit_people(text: str) -> set[str]:
prefix = text.split(":", 1)[0].strip() if ":" in text else ""
candidates = set(re.findall(r"\b(?:Dr\.\s+)?[A-ZÄÖÜ][A-Za-zÄÖÜäöüß-]+(?:\s+[A-ZÄÖÜ][A-Za-zÄÖÜäöüß-]+)*", text))
candidates.discard(prefix)
return candidates
def validate_observations(data: Any, case: dict[str, Any]) -> dict[str, Any]:
validate_case(case)
if not isinstance(data, dict):
raise ObservationValidationError("output must be an object")
_exact_keys(data, {"schema_version", "subject_id", "subject", "observations"}, "output")
if data["schema_version"] != SCHEMA_VERSION:
raise ObservationValidationError(f"schema_version must be {SCHEMA_VERSION!r}")
if data["subject_id"] != case["subject_id"] or data["subject"] != case["subject"]:
raise ObservationValidationError("model changed the fixed Discussion Subject")
observations = data["observations"]
if not isinstance(observations, list) or not observations:
raise ObservationValidationError("output.observations must be a non-empty list")
evidence = {item["evidence_id"]: item["text"] for item in case["evidence"]}
seen: set[str] = set()
for index, observation in enumerate(observations):
location = f"output.observations[{index}]"
if not isinstance(observation, dict):
raise ObservationValidationError(f"{location} must be an object")
_exact_keys(observation, OBSERVATION_KEYS, location)
observation_id = _text(observation["observation_id"], f"{location}.observation_id")
if not OBSERVATION_ID_RE.fullmatch(observation_id) or observation_id in seen:
raise ObservationValidationError(f"{location}.observation_id must be unique and match obs_N")
seen.add(observation_id)
evidence_id = _text(observation["evidence_id"], f"{location}.evidence_id")
if evidence_id not in evidence:
raise ObservationValidationError(f"{location}.evidence_id references unknown evidence: {evidence_id}")
source = evidence[evidence_id]
source_speaker = source.split(":", 1)[0].strip()
speaker = _text(observation["speaker"], f"{location}.speaker")
if speaker != source_speaker:
raise ObservationValidationError(f"{location}.speaker must match evidence speaker {source_speaker!r}")
_text(observation["content"], f"{location}.content")
explicit_people = _explicit_people(source)
for field in ("named_person", "addressee"):
person = observation[field]
if person is not None:
person = _text(person, f"{location}.{field}")
if person not in explicit_people:
raise ObservationValidationError(f"{location}.{field} is not an explicit person in evidence: {person!r}")
return data
def validate_case(case: Any) -> dict[str, Any]:
required = {"case_id", "description", "subject_id", "subject", "evidence", "semantic_requirements"}
if not isinstance(case, dict):
raise ObservationValidationError("case must be an object")
_exact_keys(case, required, "case")
for field in ("case_id", "description", "subject_id", "subject"):
_text(case[field], f"case.{field}")
if not isinstance(case["evidence"], list) or not case["evidence"]:
raise ObservationValidationError("case.evidence must be a non-empty list")
evidence_ids: set[str] = set()
for index, unit in enumerate(case["evidence"]):
_exact_keys(unit, {"evidence_id", "text"}, f"case.evidence[{index}]")
evidence_id = _text(unit["evidence_id"], f"case.evidence[{index}].evidence_id")
if evidence_id in evidence_ids:
raise ObservationValidationError(f"duplicate evidence ID: {evidence_id}")
evidence_ids.add(evidence_id)
_text(unit["text"], f"case.evidence[{index}].text")
if not isinstance(case["semantic_requirements"], list) or not case["semantic_requirements"]:
raise ObservationValidationError("case.semantic_requirements must be a non-empty list")
for index, requirement in enumerate(case["semantic_requirements"]):
_text(requirement, f"case.semantic_requirements[{index}]")
return case
def build_prompt(case: dict[str, Any]) -> str:
validate_case(case)
model_input = {key: case[key] for key in ("subject_id", "subject", "evidence")}
return PROMPT_TEMPLATE.format(input_json=json.dumps(model_input, ensure_ascii=False, indent=2))
def parse_model_json(raw_text: str) -> dict[str, Any]:
data = json.loads(raw_text)
if not isinstance(data, dict):
raise ObservationValidationError("model response JSON must be an object")
return data
def build_ollama_payload(model: str, prompt: str, num_ctx: int, num_predict: int) -> dict[str, Any]:
return {"model": model, "prompt": prompt, "think": False, "stream": False, "format": "json", "options": {"temperature": 0, "num_ctx": num_ctx, "num_predict": num_predict}}
def call_ollama(endpoint: str, model: str, prompt: str, timeout: int, num_ctx: int, num_predict: int) -> tuple[str, dict[str, Any]]:
started = time.perf_counter()
response = requests.post(endpoint, json=build_ollama_payload(model, prompt, num_ctx, num_predict), timeout=timeout)
elapsed = time.perf_counter() - started
response.raise_for_status()
body = response.json()
raw = body.get("response") if isinstance(body, dict) else None
if not isinstance(raw, str) or not raw.strip():
raise ValueError("Ollama returned no usable response text")
metadata = {"model": body.get("model", model), "elapsed_seconds": round(elapsed, 3), "total_duration_ns": body.get("total_duration"), "load_duration_ns": body.get("load_duration"), "prompt_eval_count": body.get("prompt_eval_count"), "prompt_eval_duration_ns": body.get("prompt_eval_duration"), "eval_count": body.get("eval_count"), "eval_duration_ns": body.get("eval_duration"), "configuration": {"temperature": 0, "think": False, "num_ctx": num_ctx, "num_predict": num_predict, "retries": 0}}
return raw.strip(), metadata
def load_fixture(path: Path) -> list[dict[str, Any]]:
data = json.loads(path.read_text(encoding="utf-8-sig"))
if not isinstance(data, dict) or set(data) != {"cases"} or not isinstance(data["cases"], list) or not data["cases"]:
raise ObservationValidationError("fixture must contain exactly one non-empty cases list")
seen: set[str] = set()
for case in data["cases"]:
validate_case(case)
if case["case_id"] in seen:
raise ObservationValidationError(f"duplicate case ID: {case['case_id']}")
seen.add(case["case_id"])
return data["cases"]
def _write_json(path: Path, value: Any) -> None:
path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
def run_case(case: dict[str, Any], output_root: Path, endpoint: str, model: str, timeout: int, num_ctx: int, num_predict: int) -> dict[str, Any]:
case_dir = output_root / case["case_id"]
case_dir.mkdir(parents=True, exist_ok=False)
_write_json(case_dir / "source_evidence.json", {key: case[key] for key in ("case_id", "description", "subject_id", "subject", "evidence")})
_write_json(case_dir / "gold_semantic_requirements.json", case["semantic_requirements"])
prompt = build_prompt(case)
(case_dir / "prompt.txt").write_text(prompt, encoding="utf-8")
started = time.perf_counter()
raw, metadata = call_ollama(endpoint, model, prompt, timeout, num_ctx, num_predict)
(case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8")
_write_json(case_dir / "ollama_metadata.json", metadata)
try:
parsed = parse_model_json(raw)
_write_json(case_dir / "parsed_observations.json", parsed)
validate_observations(parsed, case)
validation = {"valid": True, "error": None}
except (json.JSONDecodeError, ObservationValidationError, ValueError) as exc:
validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)}
_write_json(case_dir / "structural_validation.json", validation)
return {"case_id": case["case_id"], "structurally_valid": validation["valid"], "elapsed_seconds": round(time.perf_counter() - started, 3)}
def run_experiment(args: argparse.Namespace) -> dict[str, Any]:
cases = load_fixture(args.fixture)
args.output.mkdir(parents=True, exist_ok=False)
started = time.perf_counter()
results = []
for index, case in enumerate(cases, 1):
print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True)
results.append(run_case(case, args.output, args.endpoint, args.model, args.timeout, args.num_ctx, args.num_predict))
summary = {"experiment": "evidence_near_observation_extraction_v3", "schema_version": SCHEMA_VERSION, "model": args.model, "temperature": 0, "think": False, "retries": 0, "case_count": len(cases), "llm_call_count": len(results), "runtime_seconds": round(time.perf_counter() - started, 3), "structurally_valid_count": sum(result["structurally_valid"] for result in results), "results": results}
_write_json(args.output / "summary.json", summary)
return summary
def main() -> int:
args = parse_args()
summary = run_experiment(args)
print(json.dumps(summary, ensure_ascii=False, indent=2))
return 0 if summary["structurally_valid_count"] == summary["case_count"] else 1
if __name__ == "__main__":
raise SystemExit(main())