From 18beb3385f53c4f8cd532e3e65b6fbe475a23cd3 Mon Sep 17 00:00:00 2001 From: Martin Date: Wed, 19 Aug 2026 15:46:22 +0200 Subject: [PATCH] Add evidence-near semantic architecture experiments Record the V1-V3 experiments and accept the minimal semantic-preservation first stage. --- .gitignore | 1 + docs/experiments.md | 290 +++++++ .../run_evidence_observation_experiment.py | 16 + .../run_evidence_observation_v2_experiment.py | 16 + .../run_evidence_observation_v3_experiment.py | 16 + scripts/run_semantic_synthesis_experiment.py | 16 + .../run_topic_reconstruction_experiment.py | 16 + .../evidence_observations/__init__.py | 1 + .../evidence_observations/experiment.py | 395 ++++++++++ .../evidence_observations_v2/__init__.py | 1 + .../evidence_observations_v2/experiment.py | 344 +++++++++ .../evidence_observations_v3/__init__.py | 1 + .../evidence_observations_v3/experiment.py | 271 +++++++ .../semantic_synthesis/__init__.py | 1 + .../semantic_synthesis/experiment.py | 640 ++++++++++++++++ .../topic_reconstruction/__init__.py | 1 + .../topic_reconstruction/experiment.py | 721 ++++++++++++++++++ .../gold/evidence_observations_v1/cases.json | 147 ++++ .../gold/evidence_observations_v2/cases.json | 99 +++ .../gold/evidence_observations_v3/cases.json | 49 ++ .../semantic_synthesis_isolation/README.md | 11 + .../semantic_synthesis_isolation/cases.json | 227 ++++++ tests/gold/topic_reconstruction_v2/README.md | 20 + tests/gold/topic_reconstruction_v2/cases.json | 258 +++++++ tests/test_evidence_observation_experiment.py | 168 ++++ ...test_evidence_observation_v2_experiment.py | 160 ++++ ...test_evidence_observation_v3_experiment.py | 114 +++ tests/test_semantic_synthesis_experiment.py | 250 ++++++ tests/test_topic_reconstruction_experiment.py | 292 +++++++ 29 files changed, 4542 insertions(+) create mode 100644 scripts/run_evidence_observation_experiment.py create mode 100644 scripts/run_evidence_observation_v2_experiment.py create mode 100644 scripts/run_evidence_observation_v3_experiment.py create mode 100644 scripts/run_semantic_synthesis_experiment.py create mode 100644 scripts/run_topic_reconstruction_experiment.py create mode 100644 src/meeting_lab/evidence_observations/__init__.py create mode 100644 src/meeting_lab/evidence_observations/experiment.py create mode 100644 src/meeting_lab/evidence_observations_v2/__init__.py create mode 100644 src/meeting_lab/evidence_observations_v2/experiment.py create mode 100644 src/meeting_lab/evidence_observations_v3/__init__.py create mode 100644 src/meeting_lab/evidence_observations_v3/experiment.py create mode 100644 src/meeting_lab/semantic_synthesis/__init__.py create mode 100644 src/meeting_lab/semantic_synthesis/experiment.py create mode 100644 src/meeting_lab/topic_reconstruction/__init__.py create mode 100644 src/meeting_lab/topic_reconstruction/experiment.py create mode 100644 tests/gold/evidence_observations_v1/cases.json create mode 100644 tests/gold/evidence_observations_v2/cases.json create mode 100644 tests/gold/evidence_observations_v3/cases.json create mode 100644 tests/gold/semantic_synthesis_isolation/README.md create mode 100644 tests/gold/semantic_synthesis_isolation/cases.json create mode 100644 tests/gold/topic_reconstruction_v2/README.md create mode 100644 tests/gold/topic_reconstruction_v2/cases.json create mode 100644 tests/test_evidence_observation_experiment.py create mode 100644 tests/test_evidence_observation_v2_experiment.py create mode 100644 tests/test_evidence_observation_v3_experiment.py create mode 100644 tests/test_semantic_synthesis_experiment.py create mode 100644 tests/test_topic_reconstruction_experiment.py diff --git a/.gitignore b/.gitignore index 796fb5d..e4fadd3 100644 --- a/.gitignore +++ b/.gitignore @@ -34,6 +34,7 @@ htmlcov/ # Experiment Outputs experiments/**/output/ experiments/**/results/ +artifacts/experiments/**/ # Pipeline runtime artifacts samples/raw/ diff --git a/docs/experiments.md b/docs/experiments.md index 94b9db0..7b0d963 100644 --- a/docs/experiments.md +++ b/docs/experiments.md @@ -1350,3 +1350,293 @@ accordance with the Gold Standard methodology. Result: partially improved, not accepted as a complete BUG-015 fix. BUG-015 remains Open; no phrase-specific deterministic filter was introduced. + +## EXP-0027 — Evidence-near observation extraction + +Date: 2026-08-18 + +Hypothesis: `qwen3.5:9B` can more reliably extract evidence-near linguistic and +semantic properties than directly synthesize protocol-level events, outcomes, +actions and unresolved issues. This isolated experiment stops before semantic +interpretation and does not connect to the production pipeline. + +The fixed Gold fixture reuses the unchanged A-I evidence and fixed Discussion +Subjects from Semantic Synthesis Isolation. It defines atomic observations with +source evidence, explicit targets, a five-value relation vocabulary, modality, +temporality, evaluation, agreement, responsibility/person, uncertainty, +clarification need and free-text scope. It contains no protocol-level category +field. The validator requires sequential observation IDs, known evidence IDs, +backward-only valid observation targets, closed categorical vocabularies, +consistent responsibility/person pairs, and one canonical absence form: JSON +null for person and `absent` for scope. + +Configuration: `qwen3.5:9B`, temperature 0, `think=false`, `num_ctx=16384`, +`num_predict=4096`, no retries. All nine cases ran exactly once, for nine LLM +calls total. The run took 74.732 seconds and used 7,627 prompt-evaluation tokens +and 4,514 evaluation tokens. Exact prompts, Gold input and expectations, raw +responses, parsed observations, validation results, Ollama metadata and +comparisons are preserved under +`/tmp/meeting-lab-evidence-observations-v1-20260818/`. + +Strict automated validation/evaluation produced 0 PASS, 1 PARTIAL and 8 FAIL. +Five cases failed structure because the model represented a single target as a +one-element list, usually `["discussion_subject"]`; the accepted schema permits +a list only for two or more jointly referenced observations. Several responses +also copied the relation label `limits_scope` into the free-text scope field. +These were systematic model-output errors, not transport or parser failures. +The prompt and run were not retried or tuned. + +Human semantic review of the preserved raw responses: + +| Case | Verdict | Main result | +| --- | --- | --- | +| A | PARTIAL | Kept the geometry change uncommitted but collapsed the follow-up observation and weakened explicit uncertainty. | +| B | PARTIAL | Preserved two unselected alternatives without commitment, but used incorrect targets/relations and omitted the joint-reference observation. | +| C | FAIL | The tentative contact remained ownerless, but the follow-up was incorrectly marked factual and accepted. | +| D | PARTIAL | Preserved the negative energy consequence without creating a clarification need, but omitted the explicit uncertainty about whether washing is worthwhile and misused agreement. | +| E | PARTIAL | Preserved explicit rejection and verbal confirmation, but failed to target the confirmation at the rejection and weakened the committed future rejection to a completed fact. | +| F | FAIL | Preserved the trial-versus-series wording, but failed atomic scope targeting and incorrectly assigned responsibility to Tim from collective speech. | +| G | FAIL | Correctly recognized impersonal necessity, but promoted Martin's preference to rejection and responsibility and weakened risk/availability uncertainty. | +| H | PARTIAL | Distinguished request from commitment and captured Nina's acceptance, but named the requester as responsible in the request and failed accepted-responsibility/target encoding. | +| I | PARTIAL | Preserved the bounded production facts and the information question without assigning work, but lost scope relations and marked the unresolved permission as rejected and not uncertain. | + +Human-review total: 0 PASS, 6 PARTIAL, 3 FAIL. This review does not override +strict structural failures; it separates useful semantic signal from schema +compliance. + +Compared with Semantic Synthesis Isolation (1 PASS, 2 PARTIAL, 6 FAIL), moving +closer to evidence reduced some direct promotion behavior: the geometry mention +did not become work, the washing disadvantage did not become an unresolved +issue, both alternatives in B remained uncommitted, and the publication query +did not become an assignment. However, the important promotion errors did not +disappear. C acquired unsupported acceptance, and G still promoted a personal +preference into rejection. Positive cases were only partly preserved: explicit +rejection was recognized but incorrectly linked; trial-only language was kept +but responsibility was invented; Nina's request and commitment were recognized +but responsibility states were wrong; and the publication issue was recognized +but its uncertainty was contradicted by rejection. + +Result: **B — evidence-near extraction is promising, but specific observation +dimensions remain unreliable.** Target/relation selection, scope attachment, +responsibility state/person attribution, and agreement versus uncertainty are +not reliable enough to justify designing the later interpretation stage yet. +No production integration or later interpretation stage was implemented. + +## EXP-0028 — Evidence-Near Observation Extraction V2 + +Date: 2026-08-19 + +V2 tested whether `qwen3.5:9B` preserves the evidence needed by a later +controlled interpretation stage when direct responsibility, agreement and +semantic graph relations are removed. Responsibility was replaced by explicit +participant/discourse facts (`speaker`, `named_person`, `addressee`, singular +self-reference, collective `we`, and impersonal person reference). Agreement +was replaced by explicit affirmation, explicit negation and determination +statement signals. Graph relations were reduced to nullable scalar +`refers_to`; scope became free-text `qualifier` plus nullable scalar +`limits_target`. No later derivation stage was implemented. + +The V2 Gold fixture preserves the unchanged A-I source evidence and intended +human interpretations. It contains no responsibility, agreement, action, +decision, open-question, accepted-trial, rejected-alternative or protocol +eligibility fields. Validation enforces known evidence IDs, sequential unique +observation IDs, backward-only scalar references, closed vocabularies, boolean +participant flags, JSON-nullable participant/qualifier/reference fields and no +string `"null"`. + +Configuration: `qwen3.5:9B`, temperature 0, `think=false`, `num_ctx=16384`, +`num_predict=4096`, no retries or voting. A launch-path defect was corrected +before the live run; the failed launch made zero model calls. A sandbox-blocked +localhost attempt also made zero model calls. The completed run called the +model exactly once for each of A-I: nine calls total, in 125.645 seconds. +Persistent prompts, Gold input and expectations, raw and parsed model output, +validation, automatic comparison, Ollama metadata and human evaluation are in +`artifacts/experiments/evidence_observations_v2/20260819_v2_single_run/`. + +Strict automated comparison produced 0 PASS, 0 PARTIAL and 9 FAIL. Seven cases +were schema-invalid. The dominant serialization pattern was use of `present` +instead of the specified `explicit` for affirmation/negation; E additionally +used `none` instead of `absent` for a determination signal, while D emitted the +separate uncertainty concept as an invalid modality. These errors are +contract violations, although most `present`/`explicit` differences are +deterministically normalizable without changing meaning. A and C were valid +JSON/schema outputs but had critical semantic mismatches. + +Human semantic review: + +| Case | Verdict | Main result | +| --- | --- | --- | +| A | PARTIAL | Preserved possibility, uncertainty, atomic follow-up and its reference, but classified the initial possibility as suggestion/existing and missed implicit clarification. | +| B | PARTIAL | Preserved both alternatives without commitment or ownership, but over-fragmented, omitted references/qualifiers and added a determination signal; `present` caused schema failure. | +| C | FAIL | Preserved the initial uncertain suggestion and no ownership, but missed self-reference and again converted the follow-up possibility to a factual existing statement. | +| D | PARTIAL | Preserved possibility, explicit uncertainty, process description and negative energy consequence without assignment; references/qualifiers were lost and uncertainty was also emitted as an invalid modality. | +| E | PARTIAL | Preserved explicit no, collective speech, explicit yes and a determination statement, but missed future commitment and the confirmation reference and over-fragmented the rejection. | +| F | PARTIAL | Preserved collective speech, explicit affirmation, future test, quantity, trial-only boundary and non-adoption as series solution without individual ownership, but missed committed modality and all reference/scope attachments. | +| G | FAIL | Avoided responsibility and group-rejection promotion, but weakened risk uncertainty, personal-preference features, conditionality and impersonal necessity. | +| H | PARTIAL | Correctly preserved speaker, named addressee, request, response speaker, self-reference, affirmation and future conduct without responsibility, but duplicated the request and missed committed modality, reference and deadline qualifier. | +| I | PARTIAL | Preserved production content, information question without assignment, unresolved permission and clarification need, but lost every reference/qualifier/limit and weakened impersonal necessity. | + +Human total: 0 PASS, 7 PARTIAL, 2 FAIL. The reduced schema materially reduced +V1 promotion errors: collective speech and speaker identity no longer became +individual responsibility; personal preference no longer became a group-level +rejection field; an information question did not become work; and explicit +negation/affirmation survived as separate evidence. Useful participant evidence +also survived strongly in H and collective-speech evidence in F. + +Simplification did not make all evidence-near dimensions reliable. Scalar +references and `limits_target` were almost entirely omitted, qualifiers were +usually omitted, committed modality was missed in E, F and H, and C/G repeated +important modality, uncertainty and participant-feature errors. Some positive +semantic information therefore survived only in free-text `content`, not in +the structural signals a controlled derivation stage would need. + +Result: **B — V2 is materially better, but specific evidence-near dimensions +still require refinement.** Direct responsibility, agreement and graph-relation +classification should remain excluded. Before designing the derivation stage, +the next work should examine the minimal reliable representation of explicit +reference/scope limitation, commitment modality and participant deixis. No +production integration, Progeo run or derivation implementation was performed. + +## EXP-0029 — Evidence-Near Observation Extraction V3 — Minimal Semantic Preservation + +Date: 2026-08-19 + +Hypothesis: `qwen3.5:9B` is substantially more reliable when the first semantic +stage preserves meeting meaning as atomic natural-language observations with +provenance and only simple participant information, without classifying or +deriving higher-level meeting semantics. + +V3 uses the unchanged A-I evidence and intended meanings from V1/V2. Each +observation contains exactly `observation_id`, `evidence_id`, `content`, +`speaker`, nullable `named_person`, and nullable `addressee`. It contains no +modality, temporality, evaluation, affirmation, negation, determination, +uncertainty, clarification, responsibility, agreement, relation, reference, +qualifier, scope, limit, protocol-category or protocol-eligibility fields. +Instead, the prompt asks for conservative atomic content that retains hedges, +conditions, personal/collective/impersonal language, requests, acceptances, +rejections, quantities, deadlines and boundaries in natural language. + +Structural validation is intentionally small: exact schema keys, non-empty +observations/content, unique `obs_N` identifiers, known evidence IDs, speaker +matching its evidence, explicit named people/addressees, and no string +`"null"`. Human semantic preservation against per-case requirements is the +primary evaluation; wording differences do not fail a case. + +Configuration: `qwen3.5:9B`, temperature 0, `think=false`, `num_ctx=16384`, +`num_predict=4096`, no retries, voting or per-case tuning. One sandbox-blocked +localhost launch made zero model calls. The completed run made exactly nine +calls, one for each A-I case, in 40.074 seconds. All nine outputs passed +structural validation. Persistent source evidence, semantic requirements, +exact prompts, raw and parsed responses, validation, Ollama metadata and human +evaluation are stored under +`artifacts/experiments/evidence_observations_v3/20260819_v3_single_run/`. + +Human semantic preservation results: + +| Case | Verdict | Main result | +| --- | --- | --- | +| A | PASS | Preserved `kann`, `vielleicht`, tentative follow-up, and explicit `Dann` sequence without commitment. | +| B | PASS | Preserved insufficient strength, both alternatives, their two-approach framing, and non-selection. | +| C | PASS | Preserved Tim's tentative personal Textor contact, possible follow-up and absence of established work. | +| D | PASS | Preserved washing possibility, explicit uncertainty, process, energy consequence and absence of an invented task. | +| E | PASS | Preserved neutral cost, collective explicit rejection/non-pursuit and subsequent confirmation that it is decided. | +| F | PASS | Preserved collective possibility and test commitment, small extruder, 20 metres, next trial, trial-only limit and not-yet series adoption without individual ownership. | +| G | PARTIAL | Preserved hypothetical risk, Martin's personal stance, `wenn überhaupt`, impersonal checking need and no decision, but dropped collective `wir` from who would receive contaminated material. | +| H | PASS | Preserved Antonius's request to Nina, Friday, Nina's explicit acceptance and future first-person commitment without a responsibility field. | +| I | PASS | Preserved production/comparison boundaries, upstream effort, publication purpose, unresolved permission and clarification need without assignment. | + +Human total: 8 PASS, 1 PARTIAL, 0 FAIL. G's only material weakening changed +“that we receive contaminated material back” into an impersonal passive phrase; +the risk itself remained hypothetical. H translated `Freitag` to `Friday`, a +harmless wording difference. I retained two compound observations rather than +splitting every proposition, but all required semantic boundaries and +dependencies remained explicit. + +Compared with V2, categorical-field removal improved content preservation in +A, G and I: A retained `Dann`; G retained `wenn überhaupt`, personal `Ich` and +impersonal `Man`; I retained publication purpose and all boundaries. It also +reduced fragmentation from 38 observations in V2 to 28 in V3, with no semantic +strengthening into responsibility, group rejection, established work or +assigned clarification. F and H remain sufficiently complete in natural +language for a later interpretation experiment. No useful meaning was shown to +depend on the removed fields; the V3 content retained the useful signals that +V2's fields had attempted to encode. + +Result: **A — MINIMAL FIRST STAGE ACCEPTED.** On A-I, minimal atomic content +with evidence provenance and simple participants is sufficiently reliable to +be the candidate first semantic stage. A later bounded experiment may examine +controlled semantic interpretation, but no derivation stage, production +integration or Progeo run was implemented here. + +## EXP-0026 — Topic-oriented Discussion Subject reconstruction V2 prototype + +Date: 2026-08-11 + +Hypothesis: the primary protocol should be a topic-oriented reconstruction of +the meeting rather than a category-oriented list of extracted information. + +This first isolated prototype does not replace or connect to the production +pipeline or Working Protocol renderer. It sends small evidence-ID-tagged +transcript excerpts to `qwen3.5:9B` and requests Discussion Subjects. Each +subject may contain supported discourse events, an outcome with mandatory +scope, resulting actions and unresolved issues. Optional structures must be +omitted when absent. Every semantic object must reference known evidence IDs. + +The strict experimental schema validates: + +- non-empty subjects and globally unique semantic identifiers; +- a closed discourse-event vocabulary; +- non-empty, known and non-duplicated evidence references; +- outcome text, scope, certainty and evidence; +- action text, JSON-nullable responsibility/deadline and evidence; +- unresolved-issue text and evidence; +- omission rather than null or empty optional structures. + +Focused Gold material contains nine BUG-015/Progeo-derived cases: idea only, +multiple options, unaccepted proposal, proposal with objection, rejected +alternative, trial-scoped acceptance, no-decision discussion, resulting Action +Item, and outcome plus unresolved issue. Evaluation targets semantic identity, +development, outcome scope, actions, unresolved issues, traceability and +absence of invented commitments rather than exact wording. + +Configuration: `qwen3.5:9B`, temperature 0, `think=false`, `num_ctx=16384`, +`num_predict=4096`. Each case received exactly one model call; there were no +model retries or prompt iterations. The nine completed calls took 59.251 +seconds in aggregate and used 6,680 prompt-evaluation tokens plus 3,274 +evaluation tokens. Raw model responses, prompts, parsed JSON, metadata and +failure artifacts were preserved under +`/tmp/meeting-lab-topic-reconstruction-v2-gold-run2/` and +`/tmp/meeting-lab-topic-reconstruction-v2-gold-run3/`. Two earlier launch +attempts made zero LLM calls: one failed on the script import path and one was +blocked by sandbox networking. + +Human-reviewed results after correcting two objectively wrong Gold assumptions +without another model call: + +| Case | Verdict | Reason | +| --- | --- | --- | +| A — idea only | PARTIAL | Correct subject and no invented outcome/action, but the isolated idea was labeled `considered_option` rather than `introduced_idea`. | +| B — multiple options | FAIL | Invalid empty optional list; one discussion subject was split into three, and alternatives were promoted to tentative outcomes and invented unresolved issues. | +| C — unaccepted proposal | FAIL | Proposal was detected, but output used forbidden null/empty structures and promoted it to an Action Item. | +| D — proposal with objection | FAIL | Invalid null/empty structures; the objection was not reconstructed as a discourse event and was converted into an unresolved issue. | +| E — rejected alternative | FAIL | Rejection, scope and evidence were semantically correct, but strict validation failed on empty optional lists. | +| F — trial-only acceptance | PARTIAL | Crucially preserved the 20-metre trial scope and excluded final-series acceptance; it represented the limitation as state/unresolved context rather than a clarification event. | +| G — no decision | FAIL | Invalid empty lists, split a connected subject, represented “no decision” as a tentative outcome and invented a prerequisite outcome. | +| H — resulting action | PASS | Correct subject, explicit acceptance, Nina responsibility, Friday deadline, outcome and evidence references. | +| I — outcome plus unresolved | FAIL | Captured the production-only outcome scope, but omitted supporting evidence and the unresolved publication question; output also contained an empty optional list. | + +Result: 1 PASS, 2 PARTIAL, 6 FAIL. The most important positive signal was case +F: the model distinguished acceptance for a bounded trial from acceptance as a +final solution. It also handled the explicit action in case H well. However, +the experiment failed systematically on sparse structured output, subject +grouping and restraint around absent outcomes/actions/unresolved issues. The +model frequently mirrored optional schema fields as empty/null values, treated +alternatives as outcomes, split one discussion into multiple subjects, or +invented open issues from mere non-selection. + +The focused experiment is not promising enough to justify a real Progeo chunk +sanity check. No such run was performed, and no architecture is accepted on +the basis of this prototype. Further work should first analyze whether the +failure comes from the schema/prompt representation, the model's sparse-output +reliability, or the boundary between subject grouping and semantic synthesis. +It should not proceed through repeated prompt tuning against these nine cases. diff --git a/scripts/run_evidence_observation_experiment.py b/scripts/run_evidence_observation_experiment.py new file mode 100644 index 0000000..8b6fcbd --- /dev/null +++ b/scripts/run_evidence_observation_experiment.py @@ -0,0 +1,16 @@ +#!/usr/bin/env python3 +"""Repository entry point for the evidence-near observation experiment.""" + +import sys +from pathlib import Path + + +REPO_ROOT = Path(__file__).resolve().parents[1] +if str(REPO_ROOT) not in sys.path: + sys.path.insert(0, str(REPO_ROOT)) + +from src.meeting_lab.evidence_observations.experiment import main # noqa: E402 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_evidence_observation_v2_experiment.py b/scripts/run_evidence_observation_v2_experiment.py new file mode 100644 index 0000000..520a7ef --- /dev/null +++ b/scripts/run_evidence_observation_v2_experiment.py @@ -0,0 +1,16 @@ +#!/usr/bin/env python3 +"""Repository entry point for evidence-near observation experiment V2.""" + +import sys +from pathlib import Path + + +REPO_ROOT = Path(__file__).resolve().parents[1] +if str(REPO_ROOT) not in sys.path: + sys.path.insert(0, str(REPO_ROOT)) + +from src.meeting_lab.evidence_observations_v2.experiment import main # noqa: E402 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_evidence_observation_v3_experiment.py b/scripts/run_evidence_observation_v3_experiment.py new file mode 100644 index 0000000..a8c0bbd --- /dev/null +++ b/scripts/run_evidence_observation_v3_experiment.py @@ -0,0 +1,16 @@ +#!/usr/bin/env python3 +"""Repository entry point for evidence-near observation experiment V3.""" + +import sys +from pathlib import Path + + +REPO_ROOT = Path(__file__).resolve().parents[1] +if str(REPO_ROOT) not in sys.path: + sys.path.insert(0, str(REPO_ROOT)) + +from src.meeting_lab.evidence_observations_v3.experiment import main # noqa: E402 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_semantic_synthesis_experiment.py b/scripts/run_semantic_synthesis_experiment.py new file mode 100644 index 0000000..643dd54 --- /dev/null +++ b/scripts/run_semantic_synthesis_experiment.py @@ -0,0 +1,16 @@ +#!/usr/bin/env python3 +"""Repository entry point for the isolated semantic synthesis experiment.""" + +import sys +from pathlib import Path + + +REPO_ROOT = Path(__file__).resolve().parents[1] +if str(REPO_ROOT) not in sys.path: + sys.path.insert(0, str(REPO_ROOT)) + +from src.meeting_lab.semantic_synthesis.experiment import main # noqa: E402 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/run_topic_reconstruction_experiment.py b/scripts/run_topic_reconstruction_experiment.py new file mode 100644 index 0000000..57ec234 --- /dev/null +++ b/scripts/run_topic_reconstruction_experiment.py @@ -0,0 +1,16 @@ +#!/usr/bin/env python3 +"""Repository entry point for the isolated topic reconstruction experiment.""" + +import sys +from pathlib import Path + + +REPO_ROOT = Path(__file__).resolve().parents[1] +if str(REPO_ROOT) not in sys.path: + sys.path.insert(0, str(REPO_ROOT)) + +from src.meeting_lab.topic_reconstruction.experiment import main # noqa: E402 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/meeting_lab/evidence_observations/__init__.py b/src/meeting_lab/evidence_observations/__init__.py new file mode 100644 index 0000000..3a28284 --- /dev/null +++ b/src/meeting_lab/evidence_observations/__init__.py @@ -0,0 +1 @@ +"""Isolated evidence-near observation experiment.""" diff --git a/src/meeting_lab/evidence_observations/experiment.py b/src/meeting_lab/evidence_observations/experiment.py new file mode 100644 index 0000000..c5f6e42 --- /dev/null +++ b/src/meeting_lab/evidence_observations/experiment.py @@ -0,0 +1,395 @@ +#!/usr/bin/env python3 +"""Extract evidence-near observations for a fixed Discussion Subject.""" + +from __future__ import annotations + +import argparse +import json +import re +import time +from pathlib import Path +from typing import Any + +import requests + + +SCHEMA_VERSION = "experimental-evidence-observations-v1" +DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate" +DEFAULT_MODEL = "qwen3.5:9B" +DEFAULT_TIMEOUT = 300 +DEFAULT_NUM_CTX = 16384 +DEFAULT_NUM_PREDICT = 4096 + +RELATIONS = {"none", "supports", "opposes", "qualifies", "limits_scope"} +MODALITIES = { + "factual", + "possible", + "suggested", + "interpersonal_request", + "impersonal_necessity", + "information_question", + "committed", +} +TEMPORALITIES = {"existing", "future", "completed", "unspecified"} +EVALUATIONS = {"positive", "negative", "none"} +AGREEMENTS = {"accepted", "rejected", "unclear", "none"} +RESPONSIBILITIES = {"none", "named", "accepted"} +UNCERTAINTIES = {"present", "absent"} +CLARIFICATION_NEEDS = {"explicit", "implicit", "none"} +OBSERVATION_ID_RE = re.compile(r"^obs_[1-9][0-9]*$") + + +class ObservationValidationError(ValueError): + """Raised when an experimental fixture or model output is invalid.""" + + +PROMPT_TEMPLATE = """You extract atomic, evidence-near observations for one fixed Discussion Subject. + +Stop before protocol interpretation. Never classify anything as an idea, proposal, +objection, decision, action item, or open question. Do not determine protocol +eligibility, reconstruct topics, generate a protocol, or invent missing stages. + +Split an evidence unit into multiple observations when it directly contains multiple +propositions. Preserve every observation's source evidence ID. Use concise content in +the evidence language. + +Return exactly one JSON object with this shape: +{{ + "schema_version": "experimental-evidence-observations-v1", + "subject_id": "copy exactly", + "subject": "copy exactly", + "observations": [ + {{ + "observation_id": "obs_1", + "evidence_id": "e1", + "content": "directly supported atomic observation", + "target": "discussion_subject", + "relation": "none", + "modality": "factual", + "temporality": "existing", + "evaluation": "none", + "agreement": "none", + "responsibility": "none", + "person": null, + "uncertainty": "absent", + "clarification_need": "none", + "scope": "absent" + }} + ] +}} + +Rules: +- Number observation_id sequentially as obs_1, obs_2, ... in evidence order. +- target is "discussion_subject", one earlier observation_id, or a non-empty list of + earlier observation_ids only when the evidence jointly refers to them. +- relation is only none, supports, opposes, qualifies, or limits_scope. +- modality is only factual, possible, suggested, interpersonal_request, + impersonal_necessity, information_question, or committed. +- interpersonal_request is a direct request to another person. +- impersonal_necessity says something needs to happen without assigning it. +- information_question expresses missing information without assigning work. +- temporality is only existing, future, completed, or unspecified. +- evaluation is positive, negative, or none. Do not infer evaluation from world + knowledge. A bare cost or technical fact normally has evaluation none. +- agreement is only accepted, rejected, unclear, or none and applies to target. +- responsibility is none, named, or accepted. Use named only for an explicitly + addressed candidate and accepted only for explicit acceptance/commitment. +- person is the explicit person's name for named/accepted responsibility; otherwise + use JSON null. Mentioning or speaking in first person does not establish ownership. +- uncertainty is present or absent. +- clarification_need is explicit, implicit, or none. +- scope is an evidence-grounded qualifier, or exactly "absent". Never use null or the + string "null" anywhere. +- Confirmation of a rejection targets the rejection observation, not the option. +- A trial-only qualification targets and limits the accepted trial. +- A negative consequence can oppose another observation without requiring + clarification. +- Personal preference is not group rejection. +- Collective "we" does not name an individual owner. + +Fixed Gold input: +{input_json} +""" + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description="Run the evidence-observation Gold experiment.") + parser.add_argument("fixture", type=Path) + parser.add_argument("-o", "--output", type=Path, required=True) + parser.add_argument("--model", default=DEFAULT_MODEL) + parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT) + parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT) + parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX) + parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT) + return parser.parse_args() + + +def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None: + missing = required - value.keys() + unknown = value.keys() - required + if missing: + raise ObservationValidationError(f"{location} missing required keys: {sorted(missing)}") + if unknown: + raise ObservationValidationError(f"{location} has unknown keys: {sorted(unknown)}") + + +def _text(value: Any, location: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ObservationValidationError(f"{location} must be a non-empty string") + result = value.strip() + if result.casefold() == "null": + raise ObservationValidationError(f"{location} must not be the string 'null'") + return result + + +OBSERVATION_KEYS = { + "observation_id", "evidence_id", "content", "target", "relation", "modality", + "temporality", "evaluation", "agreement", "responsibility", "person", + "uncertainty", "clarification_need", "scope", +} + + +def _validate_target(value: Any, location: str, earlier: set[str]) -> None: + if isinstance(value, str): + target = _text(value, location) + if target != "discussion_subject" and target not in earlier: + raise ObservationValidationError(f"{location} references unknown or later observation: {target}") + return + if not isinstance(value, list) or not value: + raise ObservationValidationError(f"{location} must be discussion_subject, an earlier observation ID, or a non-empty list") + if len(value) < 2: + raise ObservationValidationError(f"{location} list must contain at least two jointly referenced observations") + seen: set[str] = set() + for index, item in enumerate(value): + target = _text(item, f"{location}[{index}]") + if target not in earlier: + raise ObservationValidationError(f"{location}[{index}] references unknown or later observation: {target}") + if target in seen: + raise ObservationValidationError(f"{location} contains duplicate target: {target}") + seen.add(target) + + +def validate_observations(data: Any, case: dict[str, Any]) -> dict[str, Any]: + validate_case(case) + if not isinstance(data, dict): + raise ObservationValidationError("output must be an object") + _exact_keys(data, {"schema_version", "subject_id", "subject", "observations"}, "output") + if data["schema_version"] != SCHEMA_VERSION: + raise ObservationValidationError(f"schema_version must be {SCHEMA_VERSION!r}") + if data["subject_id"] != case["subject_id"] or data["subject"] != case["subject"]: + raise ObservationValidationError("model changed the fixed Discussion Subject") + observations = data["observations"] + if not isinstance(observations, list) or not observations: + raise ObservationValidationError("output.observations must be a non-empty array") + known_evidence = {item["evidence_id"] for item in case["evidence"]} + earlier: set[str] = set() + for index, observation in enumerate(observations, start=1): + location = f"output.observations[{index - 1}]" + if not isinstance(observation, dict): + raise ObservationValidationError(f"{location} must be an object") + _exact_keys(observation, OBSERVATION_KEYS, location) + observation_id = _text(observation["observation_id"], f"{location}.observation_id") + if not OBSERVATION_ID_RE.fullmatch(observation_id) or observation_id != f"obs_{index}": + raise ObservationValidationError(f"{location}.observation_id must be obs_{index}") + evidence_id = _text(observation["evidence_id"], f"{location}.evidence_id") + if evidence_id not in known_evidence: + raise ObservationValidationError(f"{location}.evidence_id references unknown evidence: {evidence_id}") + _text(observation["content"], f"{location}.content") + _validate_target(observation["target"], f"{location}.target", earlier) + for field, values in ( + ("relation", RELATIONS), ("modality", MODALITIES), + ("temporality", TEMPORALITIES), ("evaluation", EVALUATIONS), + ("agreement", AGREEMENTS), ("responsibility", RESPONSIBILITIES), + ("uncertainty", UNCERTAINTIES), ("clarification_need", CLARIFICATION_NEEDS), + ): + if observation[field] not in values: + raise ObservationValidationError(f"{location}.{field} is invalid: {observation[field]!r}") + person = observation["person"] + if observation["responsibility"] == "none": + if person is not None: + raise ObservationValidationError(f"{location}.person must be JSON null when responsibility is none") + else: + _text(person, f"{location}.person") + scope = _text(observation["scope"], f"{location}.scope") + if scope.casefold() == "null": + raise ObservationValidationError(f"{location}.scope must use 'absent', not 'null'") + earlier.add(observation_id) + return data + + +def validate_case(case: Any) -> dict[str, Any]: + if not isinstance(case, dict): + raise ObservationValidationError("case must be an object") + _exact_keys(case, {"case_id", "description", "subject_id", "subject", "evidence", "expected_observations"}, "case") + for field in ("case_id", "description", "subject_id", "subject"): + _text(case[field], f"case.{field}") + evidence = case["evidence"] + if not isinstance(evidence, list) or not evidence: + raise ObservationValidationError("case.evidence must be a non-empty array") + seen: set[str] = set() + for index, unit in enumerate(evidence): + location = f"case.evidence[{index}]" + if not isinstance(unit, dict): + raise ObservationValidationError(f"{location} must be an object") + _exact_keys(unit, {"evidence_id", "text"}, location) + evidence_id = _text(unit["evidence_id"], f"{location}.evidence_id") + if evidence_id in seen: + raise ObservationValidationError(f"duplicate evidence ID: {evidence_id}") + seen.add(evidence_id) + _text(unit["text"], f"{location}.text") + expected = case["expected_observations"] + if not isinstance(expected, list) or not expected: + raise ObservationValidationError("case.expected_observations must be a non-empty array") + return case + + +def validate_fixture_case(case: dict[str, Any]) -> dict[str, Any]: + validate_case(case) + data = {"schema_version": SCHEMA_VERSION, "subject_id": case["subject_id"], "subject": case["subject"], "observations": case["expected_observations"]} + validate_observations(data, case) + return case + + +def build_prompt(case: dict[str, Any]) -> str: + validate_fixture_case(case) + model_input = {"subject_id": case["subject_id"], "subject": case["subject"], "evidence": case["evidence"]} + return PROMPT_TEMPLATE.format(input_json=json.dumps(model_input, ensure_ascii=False, indent=2)) + + +def parse_model_json(raw_text: str) -> dict[str, Any]: + data = json.loads(raw_text) + if not isinstance(data, dict): + raise ObservationValidationError("model response JSON must be an object") + return data + + +def build_ollama_payload(model: str, prompt: str, num_ctx: int, num_predict: int) -> dict[str, Any]: + return {"model": model, "prompt": prompt, "think": False, "stream": False, "format": "json", "options": {"temperature": 0, "num_ctx": num_ctx, "num_predict": num_predict}} + + +def call_ollama(endpoint: str, model: str, prompt: str, timeout: int, num_ctx: int, num_predict: int) -> tuple[str, dict[str, Any]]: + started = time.perf_counter() + response = requests.post(endpoint, json=build_ollama_payload(model, prompt, num_ctx, num_predict), timeout=timeout) + elapsed = time.perf_counter() - started + response.raise_for_status() + body = response.json() + raw_text = body.get("response") if isinstance(body, dict) else None + if not isinstance(raw_text, str) or not raw_text.strip(): + raise ValueError("Ollama returned no usable response text") + metadata = { + "model": body.get("model", model), "elapsed_seconds": round(elapsed, 3), + "total_duration_ns": body.get("total_duration"), "load_duration_ns": body.get("load_duration"), + "prompt_eval_count": body.get("prompt_eval_count"), "prompt_eval_duration_ns": body.get("prompt_eval_duration"), + "eval_count": body.get("eval_count"), "eval_duration_ns": body.get("eval_duration"), + "configuration": {"temperature": 0, "think": False, "num_ctx": num_ctx, "num_predict": num_predict, "retries": 0}, + } + return raw_text.strip(), metadata + + +COMPARE_FIELDS = ("evidence_id", "target", "relation", "modality", "temporality", "evaluation", "agreement", "responsibility", "person", "uncertainty", "clarification_need") + + +def _scope_matches(actual: str, expected: str) -> bool: + if expected == "absent": + return actual == "absent" + expected_terms = [term.strip().casefold() for term in expected.split("|")] + folded = actual.casefold() + return any(term in folded for term in expected_terms) + + +def evaluate_observations(data: dict[str, Any], expected: list[dict[str, Any]]) -> dict[str, Any]: + actual = data["observations"] + checks: list[dict[str, Any]] = [] + pair_count = min(len(actual), len(expected)) + checks.append({"name": "observation_count", "passed": len(actual) == len(expected), "critical": False}) + categories = {"missing_observations": max(0, len(expected) - len(actual)), "invented_observations": max(0, len(actual) - len(expected)), "stronger_commitment": 0, "weaker_commitment": 0, "incorrect_targets_relations": 0, "incorrect_responsibility": 0, "incorrect_uncertainty_clarification": 0} + commitment_rank = {"factual": 0, "possible": 1, "suggested": 1, "information_question": 1, "impersonal_necessity": 2, "interpersonal_request": 2, "committed": 3} + for index in range(pair_count): + got, want = actual[index], expected[index] + for field in COMPARE_FIELDS: + passed = got[field] == want[field] + checks.append({"name": f"obs_{index + 1}:{field}", "passed": passed, "critical": field in {"evidence_id", "target", "relation", "modality", "agreement", "responsibility", "person"}}) + if not passed: + if field in {"target", "relation"}: categories["incorrect_targets_relations"] += 1 + if field in {"responsibility", "person"}: categories["incorrect_responsibility"] += 1 + if field in {"uncertainty", "clarification_need"}: categories["incorrect_uncertainty_clarification"] += 1 + scope_ok = _scope_matches(got["scope"], want["scope"]) + checks.append({"name": f"obs_{index + 1}:scope", "passed": scope_ok, "critical": False}) + got_rank, want_rank = commitment_rank[got["modality"]], commitment_rank[want["modality"]] + if got_rank > want_rank or (want["agreement"] == "none" and got["agreement"] in {"accepted", "rejected"}): categories["stronger_commitment"] += 1 + if got_rank < want_rank or (want["agreement"] in {"accepted", "rejected"} and got["agreement"] == "none"): categories["weaker_commitment"] += 1 + passed_count = sum(check["passed"] for check in checks) + critical_failures = [check["name"] for check in checks if check["critical"] and not check["passed"]] + ratio = passed_count / len(checks) + if ratio == 1: + verdict = "PASS" + elif ratio >= 0.7 and categories["stronger_commitment"] == 0 and categories["incorrect_responsibility"] == 0: + verdict = "PARTIAL" + else: + verdict = "FAIL" + return {"verdict": verdict, "matched_checks": passed_count, "check_count": len(checks), "match_ratio": round(ratio, 3), "critical_failures": critical_failures, "error_categories": categories, "checks": checks} + + +def load_fixture(path: Path) -> list[dict[str, Any]]: + data = json.loads(path.read_text(encoding="utf-8-sig")) + if not isinstance(data, dict) or set(data) != {"cases"} or not isinstance(data["cases"], list) or not data["cases"]: + raise ObservationValidationError("fixture must contain exactly one non-empty cases list") + seen: set[str] = set() + for case in data["cases"]: + validate_fixture_case(case) + if case["case_id"] in seen: + raise ObservationValidationError(f"duplicate case ID: {case['case_id']}") + seen.add(case["case_id"]) + return data["cases"] + + +def _write_json(path: Path, value: Any) -> None: + path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + + +def run_case(case: dict[str, Any], output_root: Path, endpoint: str, model: str, timeout: int, num_ctx: int, num_predict: int) -> dict[str, Any]: + case_dir = output_root / case["case_id"] + case_dir.mkdir(parents=True, exist_ok=False) + _write_json(case_dir / "gold_input.json", {key: case[key] for key in ("case_id", "description", "subject_id", "subject", "evidence")}) + _write_json(case_dir / "gold_expected_observations.json", case["expected_observations"]) + prompt = build_prompt(case) + (case_dir / "prompt.txt").write_text(prompt, encoding="utf-8") + started = time.perf_counter() + try: + raw, metadata = call_ollama(endpoint, model, prompt, timeout, num_ctx, num_predict) + (case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8") + _write_json(case_dir / "ollama_metadata.json", metadata) + parsed = parse_model_json(raw) + _write_json(case_dir / "parsed_observations.json", parsed) + validate_observations(parsed, case) + validation = {"valid": True, "error": None} + evaluation = evaluate_observations(parsed, case["expected_observations"]) + except requests.RequestException: + raise + except (json.JSONDecodeError, ObservationValidationError, ValueError) as exc: + validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)} + evaluation = {"verdict": "FAIL", "matched_checks": 0, "check_count": 0, "match_ratio": 0, "critical_failures": ["schema_validation"], "error_categories": {}, "checks": []} + _write_json(case_dir / "validation_result.json", validation) + result = {"case_id": case["case_id"], **evaluation, "elapsed_seconds": round(time.perf_counter() - started, 3)} + _write_json(case_dir / "evaluation_result.json", result) + return result + + +def run_experiment(args: argparse.Namespace) -> dict[str, Any]: + cases = load_fixture(args.fixture) + args.output.mkdir(parents=True, exist_ok=False) + started = time.perf_counter() + results = [] + for index, case in enumerate(cases, start=1): + print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True) + results.append(run_case(case, args.output, args.endpoint, args.model, args.timeout, args.num_ctx, args.num_predict)) + summary = {"experiment": "evidence_near_observation_extraction", "schema_version": SCHEMA_VERSION, "model": args.model, "temperature": 0, "think": False, "retries": 0, "case_count": len(cases), "llm_call_count": len(results), "runtime_seconds": round(time.perf_counter() - started, 3), "verdict_counts": {v: sum(r["verdict"] == v for r in results) for v in ("PASS", "PARTIAL", "FAIL")}, "results": results} + _write_json(args.output / "summary.json", summary) + return summary + + +def main() -> int: + args = parse_args() + summary = run_experiment(args) + print(json.dumps(summary, ensure_ascii=False, indent=2)) + return 0 if summary["verdict_counts"]["FAIL"] == 0 else 1 diff --git a/src/meeting_lab/evidence_observations_v2/__init__.py b/src/meeting_lab/evidence_observations_v2/__init__.py new file mode 100644 index 0000000..bf24f01 --- /dev/null +++ b/src/meeting_lab/evidence_observations_v2/__init__.py @@ -0,0 +1 @@ +"""Reduced-semantic-load evidence observation experiment.""" diff --git a/src/meeting_lab/evidence_observations_v2/experiment.py b/src/meeting_lab/evidence_observations_v2/experiment.py new file mode 100644 index 0000000..5a3319e --- /dev/null +++ b/src/meeting_lab/evidence_observations_v2/experiment.py @@ -0,0 +1,344 @@ +#!/usr/bin/env python3 +"""Extract reduced-semantic-load evidence-near observations.""" + +from __future__ import annotations + +import argparse +import json +import re +import time +from pathlib import Path +from typing import Any + +import requests + + +SCHEMA_VERSION = "experimental-evidence-observations-v2" +DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate" +DEFAULT_MODEL = "qwen3.5:9B" +MODALITIES = {"factual", "possible", "suggested", "interpersonal_request", "impersonal_necessity", "information_question", "committed"} +TEMPORALITIES = {"existing", "future", "completed", "unspecified"} +EVALUATIONS = {"positive", "negative", "none"} +BINARY_SIGNALS = {"explicit", "absent"} +PRESENCE_SIGNALS = {"present", "absent"} +CLARIFICATION_NEEDS = {"explicit", "implicit", "none"} +OBSERVATION_ID_RE = re.compile(r"^obs_[1-9][0-9]*$") + + +class ObservationValidationError(ValueError): + """Raised for invalid fixtures or model output.""" + + +PROMPT_TEMPLATE = """You extract atomic linguistic and discourse observations for one fixed Discussion Subject. + +Preserve only facts directly expressed by the evidence. Do not derive responsibility, +agreement, decisions, action items, open questions, accepted trials, rejected +alternatives, established actions, or protocol eligibility. Speaker identity, a name, +an addressee, first-person language, collective "we", and impersonal "man" never by +themselves establish responsibility. + +Return exactly one JSON object with this shape: +{{ + "schema_version": "experimental-evidence-observations-v2", + "subject_id": "copy exactly", + "subject": "copy exactly", + "observations": [ + {{ + "observation_id": "obs_1", + "evidence_id": "e1", + "content": "directly supported atomic observation", + "refers_to": null, + "speaker": "name copied from evidence or null", + "named_person": null, + "addressee": null, + "self_reference": false, + "collective_we": false, + "impersonal_person_reference": false, + "modality": "factual", + "temporality": "existing", + "evaluation": "none", + "affirmation": "absent", + "negation": "absent", + "determination_statement": "absent", + "uncertainty": "absent", + "clarification_need": "none", + "qualifier": null, + "limits_target": null + }} + ] +}} + +Rules: +- Produce multiple observations for distinct propositions in one evidence unit, but do + not fragment a single proposition unnecessarily. +- observation_id is sequential in evidence order. evidence_id must be copied exactly. +- refers_to is null or one earlier observation_id when the utterance explicitly refers + to it. Never use arrays. Preserve joint-reference utterances without inventing a + multi-target graph. +- speaker is the explicit transcript speaker. named_person is a person explicitly + named in the proposition. addressee is a person explicitly addressed. +- self_reference marks singular first-person self-reference. collective_we marks + collective first-person language. impersonal_person_reference marks impersonal + person expressions such as German "man". +- modality is factual, possible, suggested, interpersonal_request, + impersonal_necessity, information_question, or committed. +- temporality is existing, future, completed, or unspecified. +- evaluation is positive, negative, or none, only when linguistically supported. +- affirmation is explicit only for an explicit affirmative discourse signal such as + "ja". negation is explicit only for directly expressed negation/rejection. +- determination_statement is present only when the utterance explicitly says a + determination has been made. +- uncertainty is present or absent. clarification_need is explicit, implicit, or none. +- qualifier is null or concise evidence-grounded qualifying text. +- limits_target is null or one earlier observation explicitly limited in validity or + scope by this observation. +- Use JSON null, never the string "null". Output no fields beyond the schema. + +Fixed Gold input: +{input_json} +""" + + +OBSERVATION_KEYS = { + "observation_id", "evidence_id", "content", "refers_to", "speaker", + "named_person", "addressee", "self_reference", "collective_we", + "impersonal_person_reference", "modality", "temporality", "evaluation", + "affirmation", "negation", "determination_statement", "uncertainty", + "clarification_need", "qualifier", "limits_target", +} + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser() + parser.add_argument("fixture", type=Path) + parser.add_argument("-o", "--output", type=Path, required=True) + parser.add_argument("--model", default=DEFAULT_MODEL) + parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT) + parser.add_argument("--timeout", type=int, default=300) + parser.add_argument("--num-ctx", type=int, default=16384) + parser.add_argument("--num-predict", type=int, default=4096) + return parser.parse_args() + + +def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None: + missing, unknown = required - value.keys(), value.keys() - required + if missing: + raise ObservationValidationError(f"{location} missing required keys: {sorted(missing)}") + if unknown: + raise ObservationValidationError(f"{location} has unknown keys: {sorted(unknown)}") + + +def _text(value: Any, location: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ObservationValidationError(f"{location} must be a non-empty string") + result = value.strip() + if result.casefold() == "null": + raise ObservationValidationError(f"{location} must not be the string 'null'") + return result + + +def _nullable_text(value: Any, location: str) -> None: + if value is not None: + _text(value, location) + + +def _prior_reference(value: Any, location: str, earlier: set[str]) -> None: + if value is None: + return + reference = _text(value, location) + if reference not in earlier: + raise ObservationValidationError(f"{location} references unknown or later observation: {reference}") + + +def validate_observations(data: Any, case: dict[str, Any]) -> dict[str, Any]: + validate_case(case) + if not isinstance(data, dict): + raise ObservationValidationError("output must be an object") + _exact_keys(data, {"schema_version", "subject_id", "subject", "observations"}, "output") + if data["schema_version"] != SCHEMA_VERSION: + raise ObservationValidationError(f"schema_version must be {SCHEMA_VERSION!r}") + if data["subject_id"] != case["subject_id"] or data["subject"] != case["subject"]: + raise ObservationValidationError("model changed the fixed Discussion Subject") + observations = data["observations"] + if not isinstance(observations, list) or not observations: + raise ObservationValidationError("output.observations must be a non-empty array") + known_evidence = {item["evidence_id"] for item in case["evidence"]} + earlier: set[str] = set() + for index, observation in enumerate(observations, 1): + location = f"output.observations[{index - 1}]" + if not isinstance(observation, dict): + raise ObservationValidationError(f"{location} must be an object") + _exact_keys(observation, OBSERVATION_KEYS, location) + observation_id = _text(observation["observation_id"], f"{location}.observation_id") + if not OBSERVATION_ID_RE.fullmatch(observation_id) or observation_id != f"obs_{index}": + raise ObservationValidationError(f"{location}.observation_id must be obs_{index}") + evidence_id = _text(observation["evidence_id"], f"{location}.evidence_id") + if evidence_id not in known_evidence: + raise ObservationValidationError(f"{location}.evidence_id references unknown evidence: {evidence_id}") + _text(observation["content"], f"{location}.content") + _prior_reference(observation["refers_to"], f"{location}.refers_to", earlier) + _prior_reference(observation["limits_target"], f"{location}.limits_target", earlier) + for field in ("speaker", "named_person", "addressee", "qualifier"): + _nullable_text(observation[field], f"{location}.{field}") + for field in ("self_reference", "collective_we", "impersonal_person_reference"): + if not isinstance(observation[field], bool): + raise ObservationValidationError(f"{location}.{field} must be boolean") + for field, values in ( + ("modality", MODALITIES), ("temporality", TEMPORALITIES), + ("evaluation", EVALUATIONS), ("affirmation", BINARY_SIGNALS), + ("negation", BINARY_SIGNALS), ("determination_statement", PRESENCE_SIGNALS), + ("uncertainty", PRESENCE_SIGNALS), ("clarification_need", CLARIFICATION_NEEDS), + ): + if observation[field] not in values: + raise ObservationValidationError(f"{location}.{field} is invalid: {observation[field]!r}") + earlier.add(observation_id) + return data + + +def validate_case(case: Any) -> dict[str, Any]: + if not isinstance(case, dict): + raise ObservationValidationError("case must be an object") + _exact_keys(case, {"case_id", "description", "subject_id", "subject", "evidence", "expected_observations"}, "case") + for field in ("case_id", "description", "subject_id", "subject"): + _text(case[field], f"case.{field}") + if not isinstance(case["evidence"], list) or not case["evidence"]: + raise ObservationValidationError("case.evidence must be a non-empty array") + seen: set[str] = set() + for index, unit in enumerate(case["evidence"]): + _exact_keys(unit, {"evidence_id", "text"}, f"case.evidence[{index}]") + evidence_id = _text(unit["evidence_id"], f"case.evidence[{index}].evidence_id") + if evidence_id in seen: + raise ObservationValidationError(f"duplicate evidence ID: {evidence_id}") + seen.add(evidence_id) + _text(unit["text"], f"case.evidence[{index}].text") + if not isinstance(case["expected_observations"], list) or not case["expected_observations"]: + raise ObservationValidationError("case.expected_observations must be a non-empty array") + return case + + +def validate_fixture_case(case: dict[str, Any]) -> dict[str, Any]: + validate_case(case) + validate_observations({"schema_version": SCHEMA_VERSION, "subject_id": case["subject_id"], "subject": case["subject"], "observations": case["expected_observations"]}, case) + return case + + +def build_prompt(case: dict[str, Any]) -> str: + validate_fixture_case(case) + model_input = {key: case[key] for key in ("subject_id", "subject", "evidence")} + return PROMPT_TEMPLATE.format(input_json=json.dumps(model_input, ensure_ascii=False, indent=2)) + + +def parse_model_json(raw_text: str) -> dict[str, Any]: + data = json.loads(raw_text) + if not isinstance(data, dict): + raise ObservationValidationError("model response JSON must be an object") + return data + + +def build_ollama_payload(model: str, prompt: str, num_ctx: int, num_predict: int) -> dict[str, Any]: + return {"model": model, "prompt": prompt, "think": False, "stream": False, "format": "json", "options": {"temperature": 0, "num_ctx": num_ctx, "num_predict": num_predict}} + + +def call_ollama(endpoint: str, model: str, prompt: str, timeout: int, num_ctx: int, num_predict: int) -> tuple[str, dict[str, Any]]: + started = time.perf_counter() + response = requests.post(endpoint, json=build_ollama_payload(model, prompt, num_ctx, num_predict), timeout=timeout) + elapsed = time.perf_counter() - started + response.raise_for_status() + body = response.json() + raw = body.get("response") if isinstance(body, dict) else None + if not isinstance(raw, str) or not raw.strip(): + raise ValueError("Ollama returned no usable response text") + metadata = {"model": body.get("model", model), "elapsed_seconds": round(elapsed, 3), "total_duration_ns": body.get("total_duration"), "load_duration_ns": body.get("load_duration"), "prompt_eval_count": body.get("prompt_eval_count"), "prompt_eval_duration_ns": body.get("prompt_eval_duration"), "eval_count": body.get("eval_count"), "eval_duration_ns": body.get("eval_duration"), "configuration": {"temperature": 0, "think": False, "num_ctx": num_ctx, "num_predict": num_predict, "retries": 0}} + return raw.strip(), metadata + + +COMPARE_FIELDS = tuple(sorted(OBSERVATION_KEYS - {"observation_id", "content", "qualifier"})) + + +def _qualifier_matches(actual: str | None, expected: str | None) -> bool: + if expected is None: + return actual is None + if actual is None: + return False + return any(term.strip().casefold() in actual.casefold() for term in expected.split("|")) + + +def evaluate_observations(data: dict[str, Any], expected: list[dict[str, Any]]) -> dict[str, Any]: + actual = data["observations"] + checks = [{"name": "observation_count", "passed": len(actual) == len(expected), "critical": False}] + for index, (got, want) in enumerate(zip(actual, expected), 1): + for field in COMPARE_FIELDS: + checks.append({"name": f"obs_{index}:{field}", "passed": got[field] == want[field], "critical": field in {"evidence_id", "refers_to", "limits_target", "modality", "affirmation", "negation", "determination_statement"}}) + checks.append({"name": f"obs_{index}:qualifier", "passed": _qualifier_matches(got["qualifier"], want["qualifier"]), "critical": False}) + passed = sum(check["passed"] for check in checks) + ratio = passed / len(checks) + critical = [check["name"] for check in checks if check["critical"] and not check["passed"]] + verdict = "PASS" if ratio == 1 else "PARTIAL" if ratio >= 0.75 and not critical else "FAIL" + return {"verdict": verdict, "matched_checks": passed, "check_count": len(checks), "match_ratio": round(ratio, 3), "critical_failures": critical, "checks": checks} + + +def load_fixture(path: Path) -> list[dict[str, Any]]: + data = json.loads(path.read_text(encoding="utf-8-sig")) + if not isinstance(data, dict) or set(data) != {"cases"} or not isinstance(data["cases"], list) or not data["cases"]: + raise ObservationValidationError("fixture must contain exactly one non-empty cases list") + seen: set[str] = set() + for case in data["cases"]: + validate_fixture_case(case) + if case["case_id"] in seen: + raise ObservationValidationError(f"duplicate case ID: {case['case_id']}") + seen.add(case["case_id"]) + return data["cases"] + + +def _write_json(path: Path, value: Any) -> None: + path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + + +def run_case(case: dict[str, Any], output_root: Path, endpoint: str, model: str, timeout: int, num_ctx: int, num_predict: int) -> dict[str, Any]: + case_dir = output_root / case["case_id"] + case_dir.mkdir(parents=True, exist_ok=False) + _write_json(case_dir / "gold_input.json", {key: case[key] for key in ("case_id", "description", "subject_id", "subject", "evidence")}) + _write_json(case_dir / "gold_expected_observations.json", case["expected_observations"]) + prompt = build_prompt(case) + (case_dir / "prompt.txt").write_text(prompt, encoding="utf-8") + started = time.perf_counter() + raw, metadata = call_ollama(endpoint, model, prompt, timeout, num_ctx, num_predict) + (case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8") + _write_json(case_dir / "ollama_metadata.json", metadata) + try: + parsed = parse_model_json(raw) + _write_json(case_dir / "parsed_observations.json", parsed) + validate_observations(parsed, case) + validation = {"valid": True, "error": None} + evaluation = evaluate_observations(parsed, case["expected_observations"]) + except (json.JSONDecodeError, ObservationValidationError, ValueError) as exc: + validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)} + evaluation = {"verdict": "FAIL", "matched_checks": 0, "check_count": 0, "match_ratio": 0, "critical_failures": ["schema_validation"], "checks": []} + _write_json(case_dir / "validation_result.json", validation) + result = {"case_id": case["case_id"], **evaluation, "elapsed_seconds": round(time.perf_counter() - started, 3)} + _write_json(case_dir / "evaluation_result.json", result) + return result + + +def run_experiment(args: argparse.Namespace) -> dict[str, Any]: + cases = load_fixture(args.fixture) + args.output.mkdir(parents=True, exist_ok=False) + started = time.perf_counter() + results = [] + for index, case in enumerate(cases, 1): + print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True) + results.append(run_case(case, args.output, args.endpoint, args.model, args.timeout, args.num_ctx, args.num_predict)) + summary = {"experiment": "evidence_near_observation_extraction_v2", "schema_version": SCHEMA_VERSION, "model": args.model, "temperature": 0, "think": False, "retries": 0, "case_count": len(cases), "llm_call_count": len(results), "runtime_seconds": round(time.perf_counter() - started, 3), "verdict_counts": {verdict: sum(result["verdict"] == verdict for result in results) for verdict in ("PASS", "PARTIAL", "FAIL")}, "results": results} + _write_json(args.output / "summary.json", summary) + return summary + + +def main() -> int: + args = parse_args() + summary = run_experiment(args) + print(json.dumps(summary, ensure_ascii=False, indent=2)) + return 0 if summary["verdict_counts"]["FAIL"] == 0 else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/meeting_lab/evidence_observations_v3/__init__.py b/src/meeting_lab/evidence_observations_v3/__init__.py new file mode 100644 index 0000000..13c2c88 --- /dev/null +++ b/src/meeting_lab/evidence_observations_v3/__init__.py @@ -0,0 +1 @@ +"""Minimal semantic-preservation observation experiment.""" diff --git a/src/meeting_lab/evidence_observations_v3/experiment.py b/src/meeting_lab/evidence_observations_v3/experiment.py new file mode 100644 index 0000000..1d8e314 --- /dev/null +++ b/src/meeting_lab/evidence_observations_v3/experiment.py @@ -0,0 +1,271 @@ +#!/usr/bin/env python3 +"""Preserve meeting meaning as minimal atomic natural-language observations.""" + +from __future__ import annotations + +import argparse +import json +import re +import time +from pathlib import Path +from typing import Any + +import requests + + +SCHEMA_VERSION = "experimental-evidence-observations-v3" +DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate" +DEFAULT_MODEL = "qwen3.5:9B" +OBSERVATION_ID_RE = re.compile(r"^obs_[1-9][0-9]*$") +OBSERVATION_KEYS = {"observation_id", "evidence_id", "content", "speaker", "named_person", "addressee"} + + +class ObservationValidationError(ValueError): + """Raised for invalid fixtures or model output.""" + + +PROMPT_TEMPLATE = """Preserve the meeting meaning in atomic natural-language observations. + +This is semantic preservation, not classification or summarization. Return only facts +faithfully contributed by the evidence. Conservative wording is more important than +elegant prose. When in doubt, preserve the source wording closely. + +Return exactly one JSON object: +{{ + "schema_version": "experimental-evidence-observations-v3", + "subject_id": "copy exactly", + "subject": "copy exactly", + "observations": [ + {{ + "observation_id": "obs_1", + "evidence_id": "e1", + "content": "atomic, semantically faithful observation", + "speaker": "speaker copied from evidence", + "named_person": null, + "addressee": null + }} + ] +}} + +Rules: +- Use only the six observation fields shown. Do not output classifications, labels, + relations, scope fields, responsibility, agreement, decisions, actions, questions, + eligibility, or any other field. +- observation_id is sequential in evidence order. Copy evidence_id and speaker. +- named_person is null or a person explicitly named in that observation's evidence. +- addressee is null or a person explicitly addressed in that observation's evidence. +- A name, speaker, or addressee never implies responsibility, acceptance, ownership, + or assignment. +- content is not a summary. Preserve distinctions needed for later interpretation: + maybe/perhaps; can/could; should/must; personal, collective, or impersonal wording; + explicit requests, acceptances, and rejections; uncertainty and unresolved status; + conditions such as "if at all"; quantities; deadlines; trial/process/comparison + boundaries; "not yet"; and sequence such as "then". +- Never strengthen modality, weaken uncertainty, turn possibility into fact, turn a + preference into group rejection, turn a request into established work, turn "we" + into individual ownership, remove conditions/limits, generalize, or invent relations. +- Split one evidence unit only when it contributes propositions that may later require + different interpretations. Do not split merely because it has several clauses. +- Do not emit observation-ID relations. When evidence clearly makes an observation + depend on the immediately preceding proposition, state that dependency naturally in + content, without inventing an antecedent. +- Preserve content in the evidence language. Use JSON null, never the string "null". + +Fixed input: +{input_json} +""" + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser() + parser.add_argument("fixture", type=Path) + parser.add_argument("-o", "--output", type=Path, required=True) + parser.add_argument("--model", default=DEFAULT_MODEL) + parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT) + parser.add_argument("--timeout", type=int, default=300) + parser.add_argument("--num-ctx", type=int, default=16384) + parser.add_argument("--num-predict", type=int, default=4096) + return parser.parse_args() + + +def _exact_keys(value: dict[str, Any], required: set[str], location: str) -> None: + missing, unknown = required - value.keys(), value.keys() - required + if missing: + raise ObservationValidationError(f"{location} missing required keys: {sorted(missing)}") + if unknown: + raise ObservationValidationError(f"{location} has unknown keys: {sorted(unknown)}") + + +def _text(value: Any, location: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ObservationValidationError(f"{location} must be a non-empty string") + result = value.strip() + if result.casefold() == "null": + raise ObservationValidationError(f"{location} must not be the string 'null'") + return result + + +def _explicit_people(text: str) -> set[str]: + prefix = text.split(":", 1)[0].strip() if ":" in text else "" + candidates = set(re.findall(r"\b(?:Dr\.\s+)?[A-ZÄÖÜ][A-Za-zÄÖÜäöüß-]+(?:\s+[A-ZÄÖÜ][A-Za-zÄÖÜäöüß-]+)*", text)) + candidates.discard(prefix) + return candidates + + +def validate_observations(data: Any, case: dict[str, Any]) -> dict[str, Any]: + validate_case(case) + if not isinstance(data, dict): + raise ObservationValidationError("output must be an object") + _exact_keys(data, {"schema_version", "subject_id", "subject", "observations"}, "output") + if data["schema_version"] != SCHEMA_VERSION: + raise ObservationValidationError(f"schema_version must be {SCHEMA_VERSION!r}") + if data["subject_id"] != case["subject_id"] or data["subject"] != case["subject"]: + raise ObservationValidationError("model changed the fixed Discussion Subject") + observations = data["observations"] + if not isinstance(observations, list) or not observations: + raise ObservationValidationError("output.observations must be a non-empty list") + evidence = {item["evidence_id"]: item["text"] for item in case["evidence"]} + seen: set[str] = set() + for index, observation in enumerate(observations): + location = f"output.observations[{index}]" + if not isinstance(observation, dict): + raise ObservationValidationError(f"{location} must be an object") + _exact_keys(observation, OBSERVATION_KEYS, location) + observation_id = _text(observation["observation_id"], f"{location}.observation_id") + if not OBSERVATION_ID_RE.fullmatch(observation_id) or observation_id in seen: + raise ObservationValidationError(f"{location}.observation_id must be unique and match obs_N") + seen.add(observation_id) + evidence_id = _text(observation["evidence_id"], f"{location}.evidence_id") + if evidence_id not in evidence: + raise ObservationValidationError(f"{location}.evidence_id references unknown evidence: {evidence_id}") + source = evidence[evidence_id] + source_speaker = source.split(":", 1)[0].strip() + speaker = _text(observation["speaker"], f"{location}.speaker") + if speaker != source_speaker: + raise ObservationValidationError(f"{location}.speaker must match evidence speaker {source_speaker!r}") + _text(observation["content"], f"{location}.content") + explicit_people = _explicit_people(source) + for field in ("named_person", "addressee"): + person = observation[field] + if person is not None: + person = _text(person, f"{location}.{field}") + if person not in explicit_people: + raise ObservationValidationError(f"{location}.{field} is not an explicit person in evidence: {person!r}") + return data + + +def validate_case(case: Any) -> dict[str, Any]: + required = {"case_id", "description", "subject_id", "subject", "evidence", "semantic_requirements"} + if not isinstance(case, dict): + raise ObservationValidationError("case must be an object") + _exact_keys(case, required, "case") + for field in ("case_id", "description", "subject_id", "subject"): + _text(case[field], f"case.{field}") + if not isinstance(case["evidence"], list) or not case["evidence"]: + raise ObservationValidationError("case.evidence must be a non-empty list") + evidence_ids: set[str] = set() + for index, unit in enumerate(case["evidence"]): + _exact_keys(unit, {"evidence_id", "text"}, f"case.evidence[{index}]") + evidence_id = _text(unit["evidence_id"], f"case.evidence[{index}].evidence_id") + if evidence_id in evidence_ids: + raise ObservationValidationError(f"duplicate evidence ID: {evidence_id}") + evidence_ids.add(evidence_id) + _text(unit["text"], f"case.evidence[{index}].text") + if not isinstance(case["semantic_requirements"], list) or not case["semantic_requirements"]: + raise ObservationValidationError("case.semantic_requirements must be a non-empty list") + for index, requirement in enumerate(case["semantic_requirements"]): + _text(requirement, f"case.semantic_requirements[{index}]") + return case + + +def build_prompt(case: dict[str, Any]) -> str: + validate_case(case) + model_input = {key: case[key] for key in ("subject_id", "subject", "evidence")} + return PROMPT_TEMPLATE.format(input_json=json.dumps(model_input, ensure_ascii=False, indent=2)) + + +def parse_model_json(raw_text: str) -> dict[str, Any]: + data = json.loads(raw_text) + if not isinstance(data, dict): + raise ObservationValidationError("model response JSON must be an object") + return data + + +def build_ollama_payload(model: str, prompt: str, num_ctx: int, num_predict: int) -> dict[str, Any]: + return {"model": model, "prompt": prompt, "think": False, "stream": False, "format": "json", "options": {"temperature": 0, "num_ctx": num_ctx, "num_predict": num_predict}} + + +def call_ollama(endpoint: str, model: str, prompt: str, timeout: int, num_ctx: int, num_predict: int) -> tuple[str, dict[str, Any]]: + started = time.perf_counter() + response = requests.post(endpoint, json=build_ollama_payload(model, prompt, num_ctx, num_predict), timeout=timeout) + elapsed = time.perf_counter() - started + response.raise_for_status() + body = response.json() + raw = body.get("response") if isinstance(body, dict) else None + if not isinstance(raw, str) or not raw.strip(): + raise ValueError("Ollama returned no usable response text") + metadata = {"model": body.get("model", model), "elapsed_seconds": round(elapsed, 3), "total_duration_ns": body.get("total_duration"), "load_duration_ns": body.get("load_duration"), "prompt_eval_count": body.get("prompt_eval_count"), "prompt_eval_duration_ns": body.get("prompt_eval_duration"), "eval_count": body.get("eval_count"), "eval_duration_ns": body.get("eval_duration"), "configuration": {"temperature": 0, "think": False, "num_ctx": num_ctx, "num_predict": num_predict, "retries": 0}} + return raw.strip(), metadata + + +def load_fixture(path: Path) -> list[dict[str, Any]]: + data = json.loads(path.read_text(encoding="utf-8-sig")) + if not isinstance(data, dict) or set(data) != {"cases"} or not isinstance(data["cases"], list) or not data["cases"]: + raise ObservationValidationError("fixture must contain exactly one non-empty cases list") + seen: set[str] = set() + for case in data["cases"]: + validate_case(case) + if case["case_id"] in seen: + raise ObservationValidationError(f"duplicate case ID: {case['case_id']}") + seen.add(case["case_id"]) + return data["cases"] + + +def _write_json(path: Path, value: Any) -> None: + path.write_text(json.dumps(value, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + + +def run_case(case: dict[str, Any], output_root: Path, endpoint: str, model: str, timeout: int, num_ctx: int, num_predict: int) -> dict[str, Any]: + case_dir = output_root / case["case_id"] + case_dir.mkdir(parents=True, exist_ok=False) + _write_json(case_dir / "source_evidence.json", {key: case[key] for key in ("case_id", "description", "subject_id", "subject", "evidence")}) + _write_json(case_dir / "gold_semantic_requirements.json", case["semantic_requirements"]) + prompt = build_prompt(case) + (case_dir / "prompt.txt").write_text(prompt, encoding="utf-8") + started = time.perf_counter() + raw, metadata = call_ollama(endpoint, model, prompt, timeout, num_ctx, num_predict) + (case_dir / "raw_model_response.txt").write_text(raw + "\n", encoding="utf-8") + _write_json(case_dir / "ollama_metadata.json", metadata) + try: + parsed = parse_model_json(raw) + _write_json(case_dir / "parsed_observations.json", parsed) + validate_observations(parsed, case) + validation = {"valid": True, "error": None} + except (json.JSONDecodeError, ObservationValidationError, ValueError) as exc: + validation = {"valid": False, "error_type": type(exc).__name__, "error": str(exc)} + _write_json(case_dir / "structural_validation.json", validation) + return {"case_id": case["case_id"], "structurally_valid": validation["valid"], "elapsed_seconds": round(time.perf_counter() - started, 3)} + + +def run_experiment(args: argparse.Namespace) -> dict[str, Any]: + cases = load_fixture(args.fixture) + args.output.mkdir(parents=True, exist_ok=False) + started = time.perf_counter() + results = [] + for index, case in enumerate(cases, 1): + print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True) + results.append(run_case(case, args.output, args.endpoint, args.model, args.timeout, args.num_ctx, args.num_predict)) + summary = {"experiment": "evidence_near_observation_extraction_v3", "schema_version": SCHEMA_VERSION, "model": args.model, "temperature": 0, "think": False, "retries": 0, "case_count": len(cases), "llm_call_count": len(results), "runtime_seconds": round(time.perf_counter() - started, 3), "structurally_valid_count": sum(result["structurally_valid"] for result in results), "results": results} + _write_json(args.output / "summary.json", summary) + return summary + + +def main() -> int: + args = parse_args() + summary = run_experiment(args) + print(json.dumps(summary, ensure_ascii=False, indent=2)) + return 0 if summary["structurally_valid_count"] == summary["case_count"] else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/meeting_lab/semantic_synthesis/__init__.py b/src/meeting_lab/semantic_synthesis/__init__.py new file mode 100644 index 0000000..3ada87f --- /dev/null +++ b/src/meeting_lab/semantic_synthesis/__init__.py @@ -0,0 +1 @@ +"""Isolated experimental semantic synthesis for known discussion subjects.""" diff --git a/src/meeting_lab/semantic_synthesis/experiment.py b/src/meeting_lab/semantic_synthesis/experiment.py new file mode 100644 index 0000000..d952c4f --- /dev/null +++ b/src/meeting_lab/semantic_synthesis/experiment.py @@ -0,0 +1,640 @@ +#!/usr/bin/env python3 +"""Run semantic synthesis with subject detection and evidence assignment fixed.""" + +from __future__ import annotations + +import argparse +import json +import time +from pathlib import Path +from typing import Any + +import requests + + +SCHEMA_VERSION = "experimental-semantic-synthesis-v1" +DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate" +DEFAULT_MODEL = "qwen3.5:9B" +DEFAULT_TIMEOUT = 300 +DEFAULT_NUM_CTX = 8192 +DEFAULT_NUM_PREDICT = 2048 + +EVENT_TYPES = { + "idea", + "option", + "proposal", + "objection", + "supporting_argument", + "clarification", + "rejection", + "scoped_acceptance", + "fact", + "technical_finding", +} +OUTCOME_STATUSES = {"established", "rejected", "scoped_acceptance", "tentative"} + + +class SynthesisValidationError(ValueError): + """Raised when isolated semantic synthesis output is structurally invalid.""" + + +PROMPT_TEMPLATE = """You perform semantic synthesis for one already known discussion subject. + +The subject boundary and evidence assignment are fixed and complete. Do not discover, +split, merge, rename, or omit the subject. Do not assign evidence to another subject. +Interpret only what the supplied evidence semantically establishes. + +Semantic distinctions: +- idea: mentioned possibility without stronger commitment +- option: alternative considered without commitment +- proposal: suggested course of action not yet established as work +- objection: argument or concern against something; not automatically unresolved +- rejection: an alternative is explicitly rejected +- scoped_acceptance: accepted only for the stated test, trial, condition, or scope +- proposal is not an action +- no decision is not a tentative decision +- mention is not an unresolved issue +- an action requires explicit assignment, acceptance, commitment, or established work +- an unresolved issue requires a concrete need explicitly left unresolved + +Preserve explicit rejection, explicit accepted work, explicit unresolved questions, +and all limits on an outcome. Never generalize trial acceptance into final acceptance. +Use only supplied evidence IDs. Keep concise semantic text in the evidence language. + +Return exactly one JSON object. Always include these fields: +{{ + "schema_version": "experimental-semantic-synthesis-v1", + "subject_id": "copy the supplied subject_id exactly", + "subject": "copy the supplied subject exactly", + "events": [ + {{ + "type": "idea|option|proposal|objection|supporting_argument|clarification|rejection|scoped_acceptance|fact|technical_finding", + "text": "supported semantic event", + "evidence_ids": ["e1"] + }} + ], + "actions": [ + {{ + "text": "established action", + "responsible": null, + "due": null, + "evidence_ids": ["e2"] + }} + ], + "unresolved_issues": [ + {{ + "text": "explicitly unresolved issue", + "evidence_ids": ["e3"] + }} + ] +}} + +The three arrays are structurally required; use [] when none exist. +Add "outcome" only when an outcome was actually established: +{{ + "status": "established|rejected|scoped_acceptance|tentative", + "text": "what was actually established", + "scope": "the exact scope, condition, or limit", + "evidence_ids": ["e2"] +}} +Omit outcome completely when there is none. Never use null for outcome. Never use the +string "null"; use JSON null only for unknown responsible or due values. + +Fixed Gold input: +{input_json} +""" + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser( + description="Run the isolated semantic-synthesis Gold experiment." + ) + parser.add_argument("fixture", type=Path, help="Fixed-subject Gold bundle JSON.") + parser.add_argument("-o", "--output", type=Path, required=True) + parser.add_argument("--model", default=DEFAULT_MODEL) + parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT) + parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT) + parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX) + parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT) + return parser.parse_args() + + +def _exact_keys( + value: dict[str, Any], required: set[str], optional: set[str], location: str +) -> None: + missing = required - value.keys() + unknown = value.keys() - required - optional + if missing: + raise SynthesisValidationError( + f"{location} missing required keys: {sorted(missing)}" + ) + if unknown: + raise SynthesisValidationError( + f"{location} has unknown keys: {sorted(unknown)}" + ) + + +def _text(value: Any, location: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise SynthesisValidationError(f"{location} must be a non-empty string") + return value.strip() + + +def validate_bundle(case: Any) -> dict[str, Any]: + if not isinstance(case, dict): + raise SynthesisValidationError("case must be an object") + _exact_keys( + case, + { + "case_id", + "description", + "subject_id", + "subject", + "evidence", + "allowed_responsible", + "expected", + }, + set(), + "case", + ) + _text(case["case_id"], "case.case_id") + _text(case["description"], "case.description") + _text(case["subject_id"], "case.subject_id") + _text(case["subject"], "case.subject") + evidence = case["evidence"] + if not isinstance(evidence, list) or not evidence: + raise SynthesisValidationError("case.evidence must be a non-empty list") + seen: set[str] = set() + for index, item in enumerate(evidence): + location = f"case.evidence[{index}]" + if not isinstance(item, dict): + raise SynthesisValidationError(f"{location} must be an object") + _exact_keys(item, {"evidence_id", "text"}, set(), location) + evidence_id = _text(item["evidence_id"], f"{location}.evidence_id") + if evidence_id in seen: + raise SynthesisValidationError(f"duplicate evidence ID: {evidence_id}") + seen.add(evidence_id) + _text(item["text"], f"{location}.text") + allowed = case["allowed_responsible"] + if not isinstance(allowed, list) or any( + not isinstance(value, str) or not value.strip() for value in allowed + ): + raise SynthesisValidationError( + "case.allowed_responsible must be a list of non-empty strings" + ) + if len(set(allowed)) != len(allowed): + raise SynthesisValidationError("case.allowed_responsible contains duplicates") + if not isinstance(case["expected"], dict): + raise SynthesisValidationError("case.expected must be an object") + return case + + +def _evidence_ids(value: Any, location: str, known: set[str]) -> list[str]: + if not isinstance(value, list) or not value: + raise SynthesisValidationError(f"{location} must be a non-empty list") + result: list[str] = [] + for index, evidence_id in enumerate(value): + evidence_id = _text(evidence_id, f"{location}[{index}]") + if evidence_id not in known: + raise SynthesisValidationError( + f"{location}[{index}] references unknown evidence ID: {evidence_id}" + ) + if evidence_id in result: + raise SynthesisValidationError( + f"{location} contains duplicate evidence ID: {evidence_id}" + ) + result.append(evidence_id) + return result + + +def _nullable_text(value: Any, location: str) -> str | None: + if value is None: + return None + result = _text(value, location) + if result.casefold() == "null": + raise SynthesisValidationError( + f"{location} must use JSON null, not the string 'null'" + ) + return result + + +def validate_synthesis(data: Any, case: dict[str, Any]) -> dict[str, Any]: + validate_bundle(case) + if not isinstance(data, dict): + raise SynthesisValidationError("output must be an object") + _exact_keys( + data, + { + "schema_version", + "subject_id", + "subject", + "events", + "actions", + "unresolved_issues", + }, + {"outcome"}, + "output", + ) + if data["schema_version"] != SCHEMA_VERSION: + raise SynthesisValidationError(f"schema_version must be {SCHEMA_VERSION!r}") + if data["subject_id"] != case["subject_id"]: + raise SynthesisValidationError("model changed fixed subject_id") + if data["subject"] != case["subject"]: + raise SynthesisValidationError("model changed fixed subject") + + known = {item["evidence_id"] for item in case["evidence"]} + events = data["events"] + if not isinstance(events, list): + raise SynthesisValidationError("output.events must be an array") + for index, event in enumerate(events): + location = f"output.events[{index}]" + if not isinstance(event, dict): + raise SynthesisValidationError(f"{location} must be an object") + _exact_keys(event, {"type", "text", "evidence_ids"}, set(), location) + if event["type"] not in EVENT_TYPES: + raise SynthesisValidationError(f"{location}.type is invalid") + _text(event["text"], f"{location}.text") + _evidence_ids(event["evidence_ids"], f"{location}.evidence_ids", known) + + if "outcome" in data: + outcome = data["outcome"] + if not isinstance(outcome, dict): + raise SynthesisValidationError( + "output.outcome must be an object when present; omit it when absent" + ) + _exact_keys( + outcome, {"status", "text", "scope", "evidence_ids"}, set(), "output.outcome" + ) + if outcome["status"] not in OUTCOME_STATUSES: + raise SynthesisValidationError("output.outcome.status is invalid") + _text(outcome["text"], "output.outcome.text") + _text(outcome["scope"], "output.outcome.scope") + _evidence_ids(outcome["evidence_ids"], "output.outcome.evidence_ids", known) + + actions = data["actions"] + if not isinstance(actions, list): + raise SynthesisValidationError("output.actions must be an array") + allowed = set(case["allowed_responsible"]) + for index, action in enumerate(actions): + location = f"output.actions[{index}]" + if not isinstance(action, dict): + raise SynthesisValidationError(f"{location} must be an object") + _exact_keys( + action, + {"text", "responsible", "due", "evidence_ids"}, + set(), + location, + ) + _text(action["text"], f"{location}.text") + responsible = _nullable_text(action["responsible"], f"{location}.responsible") + if responsible is not None and responsible not in allowed: + raise SynthesisValidationError( + f"{location}.responsible is not allowed: {responsible}" + ) + _nullable_text(action["due"], f"{location}.due") + _evidence_ids(action["evidence_ids"], f"{location}.evidence_ids", known) + + issues = data["unresolved_issues"] + if not isinstance(issues, list): + raise SynthesisValidationError("output.unresolved_issues must be an array") + for index, issue in enumerate(issues): + location = f"output.unresolved_issues[{index}]" + if not isinstance(issue, dict): + raise SynthesisValidationError(f"{location} must be an object") + _exact_keys(issue, {"text", "evidence_ids"}, set(), location) + _text(issue["text"], f"{location}.text") + _evidence_ids(issue["evidence_ids"], f"{location}.evidence_ids", known) + return data + + +def build_prompt(case: dict[str, Any]) -> str: + validate_bundle(case) + model_input = { + "subject_id": case["subject_id"], + "subject": case["subject"], + "evidence": case["evidence"], + } + return PROMPT_TEMPLATE.format( + input_json=json.dumps(model_input, ensure_ascii=False, indent=2) + ) + + +def parse_model_json(raw_text: str) -> dict[str, Any]: + data = json.loads(raw_text) + if not isinstance(data, dict): + raise SynthesisValidationError("model response JSON must be an object") + return data + + +def build_ollama_payload( + model: str, prompt: str, num_ctx: int, num_predict: int +) -> dict[str, Any]: + return { + "model": model, + "prompt": prompt, + "think": False, + "stream": False, + "format": "json", + "options": { + "temperature": 0, + "num_ctx": num_ctx, + "num_predict": num_predict, + }, + } + + +def call_ollama( + endpoint: str, + model: str, + prompt: str, + timeout: int, + num_ctx: int, + num_predict: int, +) -> tuple[str, dict[str, Any]]: + payload = build_ollama_payload(model, prompt, num_ctx, num_predict) + started = time.perf_counter() + response = requests.post(endpoint, json=payload, timeout=timeout) + elapsed = time.perf_counter() - started + response.raise_for_status() + body = response.json() + if not isinstance(body, dict): + raise ValueError("Ollama response must be an object") + raw_text = body.get("response") + if not isinstance(raw_text, str) or not raw_text.strip(): + raise ValueError("Ollama returned no usable response text") + metadata = { + "model": body.get("model", model), + "elapsed_seconds": round(elapsed, 3), + "total_duration_ns": body.get("total_duration"), + "load_duration_ns": body.get("load_duration"), + "prompt_eval_count": body.get("prompt_eval_count"), + "prompt_eval_duration_ns": body.get("prompt_eval_duration"), + "eval_count": body.get("eval_count"), + "eval_duration_ns": body.get("eval_duration"), + "configuration": { + "temperature": 0, + "think": False, + "num_ctx": num_ctx, + "num_predict": num_predict, + }, + } + return raw_text.strip(), metadata + + +def _contains(text: str, terms: list[str]) -> bool: + folded = text.casefold() + return any(term.casefold() in folded for term in terms) + + +def _refs_cover(items: list[dict[str, Any]], expected: list[str]) -> bool: + actual = { + evidence_id + for item in items + for evidence_id in item.get("evidence_ids", []) + } + return set(expected).issubset(actual) + + +def evaluate_synthesis(data: dict[str, Any], expected: dict[str, Any]) -> dict[str, Any]: + checks: list[dict[str, Any]] = [] + + def add(name: str, passed: bool, critical: bool = False) -> None: + checks.append({"name": name, "passed": passed, "critical": critical}) + + events = data["events"] + event_types = [item["type"] for item in events] + for event_type, minimum in expected.get("event_type_minimums", {}).items(): + add(f"event:{event_type}", event_types.count(event_type) >= minimum) + allowed_types = set(expected.get("allowed_event_types", EVENT_TYPES)) + add("no_unexpected_event_types", set(event_types).issubset(allowed_types)) + add( + "event_evidence", + _refs_cover(events, expected.get("event_evidence_ids", [])), + ) + + outcome_expected = expected["outcome"] + outcome = data.get("outcome") + add( + "outcome_presence", + (outcome is not None) == outcome_expected["required"], + critical=True, + ) + if outcome_expected["required"] and outcome is not None: + add("outcome_status", outcome["status"] in outcome_expected["statuses"]) + combined = f"{outcome['text']} {outcome['scope']}" + add("outcome_meaning", _contains(combined, outcome_expected["terms"])) + add( + "outcome_scope", + _contains(combined, outcome_expected["scope_terms"]), + critical=True, + ) + add( + "outcome_evidence", + set(outcome_expected["evidence_ids"]).issubset(outcome["evidence_ids"]), + critical=True, + ) + + actions = data["actions"] + expected_actions = expected["actions"] + add( + "action_count", + len(actions) == expected_actions["count"], + critical=True, + ) + if expected_actions["count"] and actions: + action_text = " ".join(item["text"] for item in actions) + add("action_meaning", _contains(action_text, expected_actions["terms"])) + if "responsible" in expected_actions: + add( + "action_responsible", + any(item["responsible"] == expected_actions["responsible"] for item in actions), + critical=True, + ) + if expected_actions.get("due_terms"): + due_text = " ".join(str(item["due"] or "") for item in actions) + add("action_due", _contains(due_text, expected_actions["due_terms"])) + add( + "action_evidence", + _refs_cover(actions, expected_actions["evidence_ids"]), + critical=True, + ) + + issues = data["unresolved_issues"] + expected_issues = expected["unresolved_issues"] + add( + "unresolved_count", + len(issues) == expected_issues["count"], + critical=True, + ) + if expected_issues["count"] and issues: + issue_text = " ".join(item["text"] for item in issues) + add("unresolved_meaning", _contains(issue_text, expected_issues["terms"])) + add( + "unresolved_evidence", + _refs_cover(issues, expected_issues["evidence_ids"]), + critical=True, + ) + + passed = sum(item["passed"] for item in checks) + critical_failures = [ + item["name"] for item in checks if item["critical"] and not item["passed"] + ] + ratio = passed / len(checks) + if ratio == 1: + verdict = "PASS" + elif ratio >= 0.7 and not critical_failures: + verdict = "PARTIAL" + else: + verdict = "FAIL" + failed = [item["name"] for item in checks if not item["passed"]] + return { + "verdict": verdict, + "reason": "All semantic checks passed." if not failed else "Failed: " + ", ".join(failed), + "passed_checks": passed, + "check_count": len(checks), + "critical_failures": critical_failures, + "checks": checks, + } + + +def load_fixture(path: Path) -> list[dict[str, Any]]: + data = json.loads(path.read_text(encoding="utf-8-sig")) + if not isinstance(data, dict) or set(data) != {"cases"}: + raise SynthesisValidationError("fixture must contain exactly a cases list") + cases = data["cases"] + if not isinstance(cases, list) or not cases: + raise SynthesisValidationError("fixture cases must be a non-empty list") + seen: set[str] = set() + for case in cases: + validate_bundle(case) + if case["case_id"] in seen: + raise SynthesisValidationError(f"duplicate case ID: {case['case_id']}") + seen.add(case["case_id"]) + return cases + + +def run_case( + case: dict[str, Any], + output_root: Path, + endpoint: str, + model: str, + timeout: int, + num_ctx: int, + num_predict: int, +) -> dict[str, Any]: + case_dir = output_root / case["case_id"] + case_dir.mkdir(parents=True, exist_ok=False) + gold_input = { + "case_id": case["case_id"], + "description": case["description"], + "subject_id": case["subject_id"], + "subject": case["subject"], + "evidence": case["evidence"], + } + (case_dir / "gold_input.json").write_text( + json.dumps(gold_input, ensure_ascii=False, indent=2) + "\n", encoding="utf-8" + ) + prompt = build_prompt(case) + (case_dir / "prompt.txt").write_text(prompt, encoding="utf-8") + started = time.perf_counter() + try: + raw_text, metadata = call_ollama( + endpoint, model, prompt, timeout, num_ctx, num_predict + ) + (case_dir / "raw_model_response.txt").write_text(raw_text + "\n", encoding="utf-8") + (case_dir / "ollama_metadata.json").write_text( + json.dumps(metadata, ensure_ascii=False, indent=2) + "\n", encoding="utf-8" + ) + parsed = parse_model_json(raw_text) + (case_dir / "parsed_response.json").write_text( + json.dumps(parsed, ensure_ascii=False, indent=2) + "\n", encoding="utf-8" + ) + validated = validate_synthesis(parsed, case) + validation = {"valid": True, "error": None} + evaluation = evaluate_synthesis(validated, case["expected"]) + except requests.RequestException: + raise + except (json.JSONDecodeError, SynthesisValidationError, ValueError) as exc: + validation = { + "valid": False, + "error_type": type(exc).__name__, + "error": str(exc), + } + evaluation = { + "verdict": "FAIL", + "reason": f"Schema validation failed: {exc}", + "passed_checks": 0, + "check_count": 0, + "critical_failures": ["schema_validation"], + "checks": [], + } + (case_dir / "validation_result.json").write_text( + json.dumps(validation, ensure_ascii=False, indent=2) + "\n", encoding="utf-8" + ) + result = { + "case_id": case["case_id"], + "description": case["description"], + **evaluation, + "elapsed_seconds": round(time.perf_counter() - started, 3), + } + (case_dir / "evaluation_result.json").write_text( + json.dumps(result, ensure_ascii=False, indent=2) + "\n", encoding="utf-8" + ) + return result + + +def run_experiment(args: argparse.Namespace) -> dict[str, Any]: + cases = load_fixture(args.fixture) + args.output.mkdir(parents=True, exist_ok=False) + started = time.perf_counter() + results: list[dict[str, Any]] = [] + for index, case in enumerate(cases, start=1): + print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True) + results.append( + run_case( + case, + args.output, + args.endpoint, + args.model, + args.timeout, + args.num_ctx, + args.num_predict, + ) + ) + summary = { + "experiment": "semantic_synthesis_isolation", + "schema_version": SCHEMA_VERSION, + "model": args.model, + "temperature": 0, + "think": False, + "num_ctx": args.num_ctx, + "num_predict": args.num_predict, + "case_count": len(cases), + "llm_call_count": len(results), + "runtime_seconds": round(time.perf_counter() - started, 3), + "verdict_counts": { + verdict: sum(result["verdict"] == verdict for result in results) + for verdict in ("PASS", "PARTIAL", "FAIL") + }, + "results": results, + } + (args.output / "summary.json").write_text( + json.dumps(summary, ensure_ascii=False, indent=2) + "\n", encoding="utf-8" + ) + return summary + + +def main() -> int: + args = parse_args() + try: + summary = run_experiment(args) + except (OSError, ValueError, requests.RequestException) as exc: + print(f"Error: {exc}") + return 1 + print(json.dumps(summary["verdict_counts"], sort_keys=True)) + print(f"Artifacts: {args.output.resolve()}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/meeting_lab/topic_reconstruction/__init__.py b/src/meeting_lab/topic_reconstruction/__init__.py new file mode 100644 index 0000000..75be9dc --- /dev/null +++ b/src/meeting_lab/topic_reconstruction/__init__.py @@ -0,0 +1 @@ +"""Experimental topic-oriented discussion reconstruction.""" diff --git a/src/meeting_lab/topic_reconstruction/experiment.py b/src/meeting_lab/topic_reconstruction/experiment.py new file mode 100644 index 0000000..a01daf7 --- /dev/null +++ b/src/meeting_lab/topic_reconstruction/experiment.py @@ -0,0 +1,721 @@ +#!/usr/bin/env python3 +"""Run an isolated Discussion Subject reconstruction experiment with Ollama.""" + +from __future__ import annotations + +import argparse +import json +import re +import time +from pathlib import Path +from typing import Any + +import requests + + +SCHEMA_VERSION = "experimental-discussion-subjects-v1" +DEFAULT_ENDPOINT = "http://127.0.0.1:11434/api/generate" +DEFAULT_MODEL = "qwen3.5:9B" +DEFAULT_TIMEOUT = 300 +DEFAULT_NUM_CTX = 16384 +DEFAULT_NUM_PREDICT = 4096 + +EVENT_TYPES = { + "introduced_idea", + "considered_option", + "proposal", + "supporting_argument", + "objection", + "clarification", + "modification", + "fact", + "technical_finding", +} +OUTCOME_CERTAINTIES = {"established", "tentative", "conditional", "rejected"} +IDENTIFIER_RE = re.compile(r"^[a-z][a-z0-9_]*$") + + +class ReconstructionValidationError(ValueError): + """Raised when experimental reconstruction output violates the schema.""" + + +PROMPT_TEMPLATE = """You reconstruct discussion subjects from meeting evidence. + +This is semantic reconstruction, not protocol writing and not flat category extraction. +Group evidence by what participants are actually discussing. For each subject, record +only supported discourse events and, when present, the actual outcome, resulting +actions, and genuinely unresolved issues. + +Important distinctions: +- discussed is not necessarily proposed +- proposed is not necessarily preferred or accepted +- preferred is not accepted +- accepted for a trial is not accepted as a final solution +- mentioned is not an unresolved question +- an outcome must preserve its scope, conditions, polarity, and uncertainty +- do not infer responsibility from mention, expertise, adjacency, or likely role +- do not invent missing stages or emit empty optional structures + +Evidence discipline: +- Use only the supplied evidence IDs in evidence_refs. +- Every subject, event, outcome, action, and unresolved issue needs at least one + evidence reference. +- Keep statements concise; do not copy long evidence passages. +- A subject may consist only of one introduced idea. + +Return one JSON object with exactly: +{{ + "schema_version": "experimental-discussion-subjects-v1", + "subjects": [ + {{ + "subject_id": "subject_1", + "title": "concise discussion subject", + "evidence_refs": ["e1"], + "development": [ + {{ + "event_id": "event_1", + "type": "introduced_idea|considered_option|proposal|supporting_argument|objection|clarification|modification|fact|technical_finding", + "text": "what happened in the discussion", + "evidence_refs": ["e1"] + }} + ], + "outcome": {{ + "text": "only what was established", + "scope": "explicit limit or full scope of the outcome", + "certainty": "established|tentative|conditional|rejected", + "evidence_refs": ["e2"] + }}, + "actions": [ + {{ + "action_id": "action_1", + "text": "established work only", + "responsible": "explicitly supported name or null", + "deadline": "explicitly supported deadline or null", + "evidence_refs": ["e3"] + }} + ], + "unresolved_issues": [ + {{ + "issue_id": "issue_1", + "text": "concrete unresolved issue", + "evidence_refs": ["e4"] + }} + ] + }} + ] +}} + +Only subject_id, title, evidence_refs are required for each subject. Omit +development, outcome, actions, or unresolved_issues when absent. Never emit null +or an empty optional list/object. + +Case ID: {case_id} +Evidence units: +{evidence_json} +""" + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser( + description="Run the isolated topic-reconstruction Gold experiment." + ) + parser.add_argument("fixture", type=Path, help="Focused Gold cases JSON.") + parser.add_argument("-o", "--output", type=Path, required=True) + parser.add_argument("--model", default=DEFAULT_MODEL) + parser.add_argument("--endpoint", default=DEFAULT_ENDPOINT) + parser.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT) + parser.add_argument("--num-ctx", type=int, default=DEFAULT_NUM_CTX) + parser.add_argument("--num-predict", type=int, default=DEFAULT_NUM_PREDICT) + parser.add_argument( + "--case", action="append", dest="case_ids", help="Run only this case ID." + ) + return parser.parse_args() + + +def _expect_exact_keys( + value: dict[str, Any], required: set[str], optional: set[str], location: str +) -> None: + missing = required - value.keys() + unknown = value.keys() - required - optional + if missing: + raise ReconstructionValidationError( + f"{location} missing required keys: {sorted(missing)}" + ) + if unknown: + raise ReconstructionValidationError( + f"{location} has unknown keys: {sorted(unknown)}" + ) + + +def _nonempty_text(value: Any, location: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ReconstructionValidationError(f"{location} must be a non-empty string") + return value.strip() + + +def _identifier(value: Any, location: str, seen: set[str]) -> str: + text = _nonempty_text(value, location) + if not IDENTIFIER_RE.fullmatch(text): + raise ReconstructionValidationError(f"{location} is not a valid identifier") + if text in seen: + raise ReconstructionValidationError(f"duplicate identifier: {text}") + seen.add(text) + return text + + +def _nullable_text(value: Any, location: str) -> str | None: + if value is None: + return None + text = _nonempty_text(value, location) + if text.casefold() == "null": + raise ReconstructionValidationError( + f"{location} must use JSON null, not the string 'null'" + ) + return text + + +def _evidence_refs(value: Any, location: str, known: set[str]) -> list[str]: + if not isinstance(value, list) or not value: + raise ReconstructionValidationError(f"{location} must be a non-empty list") + refs: list[str] = [] + for index, ref in enumerate(value): + ref = _nonempty_text(ref, f"{location}[{index}]") + if ref not in known: + raise ReconstructionValidationError( + f"{location}[{index}] references unknown evidence ID: {ref}" + ) + if ref in refs: + raise ReconstructionValidationError( + f"{location} contains duplicate evidence reference: {ref}" + ) + refs.append(ref) + return refs + + +def validate_evidence_units(evidence_units: Any) -> set[str]: + if not isinstance(evidence_units, list) or not evidence_units: + raise ReconstructionValidationError("evidence_units must be a non-empty list") + known: set[str] = set() + for index, unit in enumerate(evidence_units): + location = f"evidence_units[{index}]" + if not isinstance(unit, dict): + raise ReconstructionValidationError(f"{location} must be an object") + _expect_exact_keys(unit, {"evidence_id", "text"}, set(), location) + evidence_id = _nonempty_text(unit["evidence_id"], f"{location}.evidence_id") + if evidence_id in known: + raise ReconstructionValidationError( + f"duplicate input evidence identifier: {evidence_id}" + ) + known.add(evidence_id) + _nonempty_text(unit["text"], f"{location}.text") + return known + + +def validate_reconstruction(data: Any, evidence_units: Any) -> dict[str, Any]: + known = validate_evidence_units(evidence_units) + if not isinstance(data, dict): + raise ReconstructionValidationError("model output must be an object") + _expect_exact_keys(data, {"schema_version", "subjects"}, set(), "output") + if data["schema_version"] != SCHEMA_VERSION: + raise ReconstructionValidationError( + f"schema_version must be {SCHEMA_VERSION!r}" + ) + subjects = data["subjects"] + if not isinstance(subjects, list) or not subjects: + raise ReconstructionValidationError("subjects must be a non-empty list") + + seen: set[str] = set() + for subject_index, subject in enumerate(subjects): + location = f"subjects[{subject_index}]" + if not isinstance(subject, dict): + raise ReconstructionValidationError(f"{location} must be an object") + _expect_exact_keys( + subject, + {"subject_id", "title", "evidence_refs"}, + {"development", "outcome", "actions", "unresolved_issues"}, + location, + ) + _identifier(subject["subject_id"], f"{location}.subject_id", seen) + _nonempty_text(subject["title"], f"{location}.title") + _evidence_refs(subject["evidence_refs"], f"{location}.evidence_refs", known) + + if "development" in subject: + events = subject["development"] + if not isinstance(events, list) or not events: + raise ReconstructionValidationError( + f"{location}.development must be a non-empty list when present" + ) + for event_index, event in enumerate(events): + event_location = f"{location}.development[{event_index}]" + if not isinstance(event, dict): + raise ReconstructionValidationError( + f"{event_location} must be an object" + ) + _expect_exact_keys( + event, + {"event_id", "type", "text", "evidence_refs"}, + set(), + event_location, + ) + _identifier(event["event_id"], f"{event_location}.event_id", seen) + if event["type"] not in EVENT_TYPES: + raise ReconstructionValidationError( + f"{event_location}.type is invalid: {event['type']!r}" + ) + _nonempty_text(event["text"], f"{event_location}.text") + _evidence_refs( + event["evidence_refs"], f"{event_location}.evidence_refs", known + ) + + if "outcome" in subject: + outcome = subject["outcome"] + outcome_location = f"{location}.outcome" + if not isinstance(outcome, dict): + raise ReconstructionValidationError( + f"{outcome_location} must be a non-empty object when present" + ) + _expect_exact_keys( + outcome, + {"text", "scope", "certainty", "evidence_refs"}, + set(), + outcome_location, + ) + _nonempty_text(outcome["text"], f"{outcome_location}.text") + _nonempty_text(outcome["scope"], f"{outcome_location}.scope") + if outcome["certainty"] not in OUTCOME_CERTAINTIES: + raise ReconstructionValidationError( + f"{outcome_location}.certainty is invalid: {outcome['certainty']!r}" + ) + _evidence_refs( + outcome["evidence_refs"], f"{outcome_location}.evidence_refs", known + ) + + if "actions" in subject: + actions = subject["actions"] + if not isinstance(actions, list) or not actions: + raise ReconstructionValidationError( + f"{location}.actions must be a non-empty list when present" + ) + for action_index, action in enumerate(actions): + action_location = f"{location}.actions[{action_index}]" + if not isinstance(action, dict): + raise ReconstructionValidationError( + f"{action_location} must be an object" + ) + _expect_exact_keys( + action, + {"action_id", "text", "responsible", "deadline", "evidence_refs"}, + set(), + action_location, + ) + _identifier(action["action_id"], f"{action_location}.action_id", seen) + _nonempty_text(action["text"], f"{action_location}.text") + for field in ("responsible", "deadline"): + _nullable_text(action[field], f"{action_location}.{field}") + _evidence_refs( + action["evidence_refs"], f"{action_location}.evidence_refs", known + ) + + if "unresolved_issues" in subject: + issues = subject["unresolved_issues"] + if not isinstance(issues, list) or not issues: + raise ReconstructionValidationError( + f"{location}.unresolved_issues must be a non-empty list when present" + ) + for issue_index, issue in enumerate(issues): + issue_location = f"{location}.unresolved_issues[{issue_index}]" + if not isinstance(issue, dict): + raise ReconstructionValidationError( + f"{issue_location} must be an object" + ) + _expect_exact_keys( + issue, + {"issue_id", "text", "evidence_refs"}, + set(), + issue_location, + ) + _identifier(issue["issue_id"], f"{issue_location}.issue_id", seen) + _nonempty_text(issue["text"], f"{issue_location}.text") + _evidence_refs( + issue["evidence_refs"], f"{issue_location}.evidence_refs", known + ) + + return data + + +def build_prompt(case: dict[str, Any]) -> str: + evidence_units = case["evidence_units"] + validate_evidence_units(evidence_units) + return PROMPT_TEMPLATE.format( + case_id=case["case_id"], + evidence_json=json.dumps(evidence_units, ensure_ascii=False, indent=2), + ) + + +def parse_model_json(raw_text: str) -> dict[str, Any]: + data = json.loads(raw_text) + if not isinstance(data, dict): + raise ReconstructionValidationError("model response JSON must be an object") + return data + + +def build_ollama_payload( + model: str, prompt: str, num_ctx: int, num_predict: int +) -> dict[str, Any]: + return { + "model": model, + "prompt": prompt, + "think": False, + "stream": False, + "format": "json", + "options": { + "temperature": 0, + "num_ctx": num_ctx, + "num_predict": num_predict, + }, + } + + +def call_ollama( + endpoint: str, + model: str, + prompt: str, + timeout: int, + num_ctx: int, + num_predict: int, +) -> tuple[str, dict[str, Any]]: + payload = build_ollama_payload(model, prompt, num_ctx, num_predict) + started = time.perf_counter() + response = requests.post(endpoint, json=payload, timeout=timeout) + elapsed = time.perf_counter() - started + response.raise_for_status() + data = response.json() + if not isinstance(data, dict): + raise ValueError("Ollama response must be a JSON object") + raw_text = data.get("response") + if not isinstance(raw_text, str) or not raw_text.strip(): + raise ValueError("Ollama returned no usable response text") + metadata = { + "model": data.get("model", model), + "elapsed_seconds": round(elapsed, 3), + "total_duration_ns": data.get("total_duration"), + "load_duration_ns": data.get("load_duration"), + "prompt_eval_count": data.get("prompt_eval_count"), + "prompt_eval_duration_ns": data.get("prompt_eval_duration"), + "eval_count": data.get("eval_count"), + "eval_duration_ns": data.get("eval_duration"), + "configuration": { + "temperature": 0, + "think": False, + "num_ctx": num_ctx, + "num_predict": num_predict, + }, + } + return raw_text.strip(), metadata + + +def _all_text(subjects: list[dict[str, Any]]) -> str: + parts: list[str] = [] + for subject in subjects: + parts.append(subject["title"]) + for event in subject.get("development", []): + parts.append(event["text"]) + outcome = subject.get("outcome") + if outcome: + parts.extend((outcome["text"], outcome["scope"])) + for action in subject.get("actions", []): + parts.append(action["text"]) + for issue in subject.get("unresolved_issues", []): + parts.append(issue["text"]) + return " ".join(parts).casefold() + + +def _contains_any(text: str, terms: list[str]) -> bool: + return any(term.casefold() in text for term in terms) + + +def evaluate_reconstruction( + reconstruction: dict[str, Any], expected: dict[str, Any] +) -> dict[str, Any]: + subjects = reconstruction["subjects"] + combined = _all_text(subjects) + events = [event for subject in subjects for event in subject.get("development", [])] + outcomes = [subject["outcome"] for subject in subjects if "outcome" in subject] + actions = [action for subject in subjects for action in subject.get("actions", [])] + issues = [issue for subject in subjects for issue in subject.get("unresolved_issues", [])] + checks: list[dict[str, Any]] = [] + + def add(name: str, passed: bool, critical: bool = False) -> None: + checks.append({"name": name, "passed": passed, "critical": critical}) + + add("subject_count", len(subjects) == expected.get("subject_count", 1)) + add("subject_identity", _contains_any(combined, expected["subject_terms"])) + + event_types = {event["type"] for event in events} + for event_type in expected.get("required_event_types", []): + add(f"event_type:{event_type}", event_type in event_types) + + expected_outcome = expected.get("outcome", {}) + outcome_required = expected_outcome.get("required", False) + add( + "outcome_presence", + bool(outcomes) is outcome_required, + critical=not outcome_required and bool(outcomes), + ) + if outcome_required and outcomes: + outcome_text = " ".join( + f"{item['text']} {item['scope']}" for item in outcomes + ).casefold() + add("outcome_meaning", _contains_any(outcome_text, expected_outcome["terms"])) + add( + "outcome_scope", + _contains_any(outcome_text, expected_outcome.get("scope_terms", [])), + critical=True, + ) + add( + "outcome_certainty", + any( + item["certainty"] in expected_outcome.get("certainties", []) + for item in outcomes + ), + ) + + expected_actions = expected.get("actions", {}) + minimum_actions = expected_actions.get("minimum", 0) + add( + "action_count", + len(actions) >= minimum_actions if minimum_actions else not actions, + critical=minimum_actions == 0 and bool(actions), + ) + if minimum_actions and actions: + action_text = " ".join(item["text"] for item in actions).casefold() + add("action_meaning", _contains_any(action_text, expected_actions["terms"])) + if "responsible" in expected_actions: + add( + "action_responsibility", + any( + item["responsible"] == expected_actions["responsible"] + for item in actions + ), + critical=True, + ) + + expected_issues = expected.get("unresolved", {}) + minimum_issues = expected_issues.get("minimum", 0) + add( + "unresolved_count", + len(issues) >= minimum_issues if minimum_issues else not issues, + critical=minimum_issues == 0 and bool(issues), + ) + if minimum_issues and issues: + issue_text = " ".join(item["text"] for item in issues).casefold() + add("unresolved_meaning", _contains_any(issue_text, expected_issues["terms"])) + + passed = sum(check["passed"] for check in checks) + critical_failures = [ + check["name"] for check in checks if check["critical"] and not check["passed"] + ] + ratio = passed / len(checks) + if ratio == 1: + verdict = "PASS" + elif ratio >= 0.6 and not critical_failures: + verdict = "PARTIAL" + else: + verdict = "FAIL" + failed = [check["name"] for check in checks if not check["passed"]] + reason = "All semantic checks passed." if not failed else "Failed: " + ", ".join(failed) + return { + "verdict": verdict, + "reason": reason, + "passed_checks": passed, + "check_count": len(checks), + "critical_failures": critical_failures, + "checks": checks, + } + + +def load_fixture(path: Path) -> list[dict[str, Any]]: + data = json.loads(path.read_text(encoding="utf-8-sig")) + if not isinstance(data, dict) or set(data) != {"cases"}: + raise ValueError("fixture must contain exactly one 'cases' list") + cases = data["cases"] + if not isinstance(cases, list) or not cases: + raise ValueError("fixture cases must be a non-empty list") + seen: set[str] = set() + for index, case in enumerate(cases): + if not isinstance(case, dict): + raise ValueError(f"cases[{index}] must be an object") + required = {"case_id", "description", "evidence_units", "expected"} + if set(case) != required: + raise ValueError(f"cases[{index}] must contain exactly {sorted(required)}") + case_id = _nonempty_text(case["case_id"], f"cases[{index}].case_id") + if case_id in seen: + raise ValueError(f"duplicate case_id: {case_id}") + seen.add(case_id) + _nonempty_text(case["description"], f"cases[{index}].description") + validate_evidence_units(case["evidence_units"]) + if not isinstance(case["expected"], dict): + raise ValueError(f"cases[{index}].expected must be an object") + return cases + + +def run_case( + case: dict[str, Any], + output_root: Path, + endpoint: str, + model: str, + timeout: int, + num_ctx: int, + num_predict: int, +) -> dict[str, Any]: + case_dir = output_root / case["case_id"] + case_dir.mkdir(parents=True, exist_ok=False) + input_payload = { + "case_id": case["case_id"], + "description": case["description"], + "evidence_units": case["evidence_units"], + } + (case_dir / "input.json").write_text( + json.dumps(input_payload, ensure_ascii=False, indent=2) + "\n", + encoding="utf-8", + ) + prompt = build_prompt(case) + (case_dir / "prompt.txt").write_text(prompt, encoding="utf-8") + + started = time.perf_counter() + try: + raw_text, metadata = call_ollama( + endpoint, model, prompt, timeout, num_ctx, num_predict + ) + (case_dir / "raw_model_response.txt").write_text( + raw_text + "\n", encoding="utf-8" + ) + (case_dir / "ollama_metadata.json").write_text( + json.dumps(metadata, ensure_ascii=False, indent=2) + "\n", + encoding="utf-8", + ) + parsed = parse_model_json(raw_text) + (case_dir / "parsed_output.json").write_text( + json.dumps(parsed, ensure_ascii=False, indent=2) + "\n", + encoding="utf-8", + ) + validated = validate_reconstruction(parsed, case["evidence_units"]) + evaluation = evaluate_reconstruction(validated, case["expected"]) + except requests.RequestException as exc: + failure = { + "case_id": case["case_id"], + "error_type": type(exc).__name__, + "error": str(exc), + "elapsed_seconds": round(time.perf_counter() - started, 3), + } + (case_dir / "validation_failure.json").write_text( + json.dumps(failure, ensure_ascii=False, indent=2) + "\n", + encoding="utf-8", + ) + raise + except (json.JSONDecodeError, ReconstructionValidationError, ValueError) as exc: + elapsed = round(time.perf_counter() - started, 3) + failure = { + "case_id": case["case_id"], + "error_type": type(exc).__name__, + "error": str(exc), + "elapsed_seconds": elapsed, + } + (case_dir / "validation_failure.json").write_text( + json.dumps(failure, ensure_ascii=False, indent=2) + "\n", + encoding="utf-8", + ) + result = { + "case_id": case["case_id"], + "description": case["description"], + "verdict": "FAIL", + "reason": f"{type(exc).__name__}: {exc}", + "passed_checks": 0, + "check_count": 0, + "critical_failures": ["schema_validation"], + "checks": [], + "elapsed_seconds": elapsed, + "subject_titles": [], + } + (case_dir / "evaluation.json").write_text( + json.dumps(result, ensure_ascii=False, indent=2) + "\n", + encoding="utf-8", + ) + return result + + result = { + "case_id": case["case_id"], + "description": case["description"], + **evaluation, + "elapsed_seconds": metadata["elapsed_seconds"], + "subject_titles": [item["title"] for item in validated["subjects"]], + } + (case_dir / "evaluation.json").write_text( + json.dumps(result, ensure_ascii=False, indent=2) + "\n", + encoding="utf-8", + ) + return result + + +def run_experiment(args: argparse.Namespace) -> dict[str, Any]: + cases = load_fixture(args.fixture) + selected = set(args.case_ids or []) + if selected: + known = {case["case_id"] for case in cases} + unknown = selected - known + if unknown: + raise ValueError(f"unknown requested case IDs: {sorted(unknown)}") + cases = [case for case in cases if case["case_id"] in selected] + + args.output.mkdir(parents=True, exist_ok=False) + results: list[dict[str, Any]] = [] + started = time.perf_counter() + for index, case in enumerate(cases, start=1): + print(f"[{index}/{len(cases)}] {case['case_id']}", flush=True) + results.append( + run_case( + case, + args.output, + args.endpoint, + args.model, + args.timeout, + args.num_ctx, + args.num_predict, + ) + ) + summary = { + "experiment": "topic_reconstruction_v2", + "schema_version": SCHEMA_VERSION, + "model": args.model, + "temperature": 0, + "think": False, + "case_count": len(cases), + "llm_call_count": len(results), + "runtime_seconds": round(time.perf_counter() - started, 3), + "verdict_counts": { + verdict: sum(item["verdict"] == verdict for item in results) + for verdict in ("PASS", "PARTIAL", "FAIL") + }, + "results": results, + } + (args.output / "summary.json").write_text( + json.dumps(summary, ensure_ascii=False, indent=2) + "\n", + encoding="utf-8", + ) + return summary + + +def main() -> int: + args = parse_args() + try: + summary = run_experiment(args) + except (OSError, ValueError, requests.RequestException) as exc: + print(f"Error: {exc}") + return 1 + print(json.dumps(summary["verdict_counts"], sort_keys=True)) + print(f"Artifacts: {args.output.resolve()}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/gold/evidence_observations_v1/cases.json b/tests/gold/evidence_observations_v1/cases.json new file mode 100644 index 0000000..50bc63a --- /dev/null +++ b/tests/gold/evidence_observations_v1/cases.json @@ -0,0 +1,147 @@ +{ + "cases": [ + { + "case_id": "a_idea_only", + "description": "Possible geometry optimization without commitment.", + "subject_id": "subject_a", + "subject": "Optimierung der Geometrie", + "evidence": [{"evidence_id": "e1", "text": "Martin: Die Geometrie kann man vielleicht noch optimieren. Dann würde man mal gucken, was herauskommt."}], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Die Geometrie kann vielleicht optimiert werden.","target":"discussion_subject","relation":"none","modality":"possible","temporality":"future","evaluation":"positive","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"none","scope":"absent"}, + {"observation_id":"obs_2","evidence_id":"e1","content":"Danach könnte betrachtet werden, was herauskommt.","target":"obs_1","relation":"qualifies","modality":"suggested","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"implicit","scope":"nach der Optimierung|danach"} + ] + }, + { + "case_id": "b_multiple_options", + "description": "Two alternatives for insufficient grid strength.", + "subject_id": "subject_b", + "subject": "Umgang mit unzureichender Festigkeit des 40-40-Gitters", + "evidence": [ + {"evidence_id":"e1","text":"Martin: Die Festigkeit reicht für das 40-40-Gitter noch nicht aus."}, + {"evidence_id":"e2","text":"Martin: Man könnte mehr Masse für die gleiche Festigkeit einsetzen."}, + {"evidence_id":"e3","text":"Martin: Oder wir verkaufen es nicht als 40-40-Gitter, sondern machen ein 20-20 daraus. Das wären die zwei Ansätze."} + ], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Die Festigkeit reicht noch nicht aus.","target":"discussion_subject","relation":"none","modality":"factual","temporality":"existing","evaluation":"negative","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"40-40-Gitter|40-40"}, + {"observation_id":"obs_2","evidence_id":"e2","content":"Mehr Masse könnte für die gleiche Festigkeit eingesetzt werden.","target":"obs_1","relation":"qualifies","modality":"possible","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"mehr Masse|gleiche Festigkeit"}, + {"observation_id":"obs_3","evidence_id":"e3","content":"Das Produkt könnte als 20-20 statt 40-40 ausgeführt werden.","target":"obs_1","relation":"qualifies","modality":"suggested","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"20-20|statt 40-40"}, + {"observation_id":"obs_4","evidence_id":"e3","content":"Die vorherigen Möglichkeiten sind die zwei Ansätze.","target":["obs_2","obs_3"],"relation":"qualifies","modality":"factual","temporality":"existing","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"absent"} + ] + }, + { + "case_id": "c_unaccepted_proposal", + "description": "Suggested Textor contact without established work.", + "subject_id": "subject_c", + "subject": "Erneute Kontaktaufnahme mit Dirk Textor zur Einschätzung", + "evidence": [ + {"evidence_id":"e1","text":"Tim: Ich würde vielleicht Dirk Textor noch einmal kontaktieren und fragen, wie er das einschätzt."}, + {"evidence_id":"e2","text":"Tim: Das kann man ja mit ihm einfach noch einmal rückkoppeln."} + ], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Tim erwägt, Dirk Textor erneut zu kontaktieren und nach seiner Einschätzung zu fragen.","target":"discussion_subject","relation":"none","modality":"suggested","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"none","scope":"Dirk Textors Einschätzung|erneut kontaktieren"}, + {"observation_id":"obs_2","evidence_id":"e2","content":"Eine erneute Rückkopplung mit Dirk Textor ist möglich.","target":"obs_1","relation":"supports","modality":"possible","temporality":"future","evaluation":"positive","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"Rückkopplung mit Dirk Textor|mit ihm"} + ] + }, + { + "case_id": "d_proposal_with_objection", + "description": "Washing possibility and explicit energy disadvantage.", + "subject_id": "subject_d", + "subject": "Waschen des Materials vor der weiteren Verarbeitung", + "evidence": [ + {"evidence_id":"e1","text":"Antonius: Man könnte das Material vor der weiteren Verarbeitung waschen."}, + {"evidence_id":"e2","text":"Martin: Ob sich das lohnt, weiß ich nicht. Waschen heißt nass machen und wieder trocknen; das ist ein wahnsinniger Energieaufwand."} + ], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Das Material könnte gewaschen werden.","target":"discussion_subject","relation":"none","modality":"possible","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"vor der weiteren Verarbeitung"}, + {"observation_id":"obs_2","evidence_id":"e2","content":"Martin weiß nicht, ob sich das Waschen lohnt.","target":"obs_1","relation":"qualifies","modality":"factual","temporality":"existing","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"none","scope":"Nutzen des Waschens|ob es sich lohnt"}, + {"observation_id":"obs_3","evidence_id":"e2","content":"Waschen umfasst Nassmachen und erneutes Trocknen.","target":"obs_1","relation":"qualifies","modality":"factual","temporality":"existing","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"Waschprozess|Nassmachen und Trocknen"}, + {"observation_id":"obs_4","evidence_id":"e2","content":"Waschen und Trocknen verursachen einen sehr hohen Energieaufwand.","target":"obs_1","relation":"opposes","modality":"factual","temporality":"existing","evaluation":"negative","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"Energieaufwand des Waschens|Waschen und Trocknen"} + ] + }, + { + "case_id": "e_rejected_alternative", + "description": "Explicit rejection followed by confirmation of that rejection.", + "subject_id": "subject_e", + "subject": "Zusammenarbeit mit Dr. Schlummer für Versuche", + "evidence": [ + {"evidence_id":"e1","text":"Antonius: Das Angebot von Dr. Schlummer für die Versuche kostet 30.000 Euro."}, + {"evidence_id":"e2","text":"Tim: Dann haben wir gesagt: Nein, die Zusammenarbeit mit Dr. Schlummer machen wir nicht."}, + {"evidence_id":"e3","text":"Antonius: Ja, das ist entschieden."} + ], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Das Angebot kostet 30.000 Euro.","target":"discussion_subject","relation":"none","modality":"factual","temporality":"existing","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"Angebot für die Versuche|30.000 Euro"}, + {"observation_id":"obs_2","evidence_id":"e2","content":"Die Zusammenarbeit mit Dr. Schlummer wird nicht durchgeführt.","target":"discussion_subject","relation":"none","modality":"committed","temporality":"future","evaluation":"none","agreement":"rejected","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"Zusammenarbeit für die Versuche|Dr. Schlummer"}, + {"observation_id":"obs_3","evidence_id":"e3","content":"Die vorherige Ablehnung ist entschieden.","target":"obs_2","relation":"supports","modality":"factual","temporality":"completed","evaluation":"none","agreement":"accepted","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"absent"} + ] + }, + { + "case_id": "f_trial_only_acceptance", + "description": "Acceptance limited to a 20-metre trial.", + "subject_id": "subject_f", + "subject": "20-Prozent-Variante im Versuch am kleinen Extruder", + "evidence": [ + {"evidence_id":"e1","text":"Martin: Wir könnten die 20-Prozent-Variante am kleinen Extruder nachstellen."}, + {"evidence_id":"e2","text":"Tim: Ja, wir testen 20 Meter dieser Variante beim nächsten Versuch."}, + {"evidence_id":"e3","text":"Tim: Das ist nur ein Versuch; damit ist die Variante noch nicht als Serienlösung festgelegt."} + ], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Die 20-Prozent-Variante könnte am kleinen Extruder nachgestellt werden.","target":"discussion_subject","relation":"none","modality":"possible","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"kleiner Extruder"}, + {"observation_id":"obs_2","evidence_id":"e2","content":"20 Meter der Variante werden beim nächsten Versuch getestet.","target":"obs_1","relation":"supports","modality":"committed","temporality":"future","evaluation":"none","agreement":"accepted","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"20 Meter beim nächsten Versuch|20 Meter"}, + {"observation_id":"obs_3","evidence_id":"e3","content":"Die Zusage gilt nur für einen Versuch.","target":"obs_2","relation":"limits_scope","modality":"factual","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"nur ein Versuch|Versuch"}, + {"observation_id":"obs_4","evidence_id":"e3","content":"Die Variante ist noch nicht als Serienlösung festgelegt.","target":"discussion_subject","relation":"qualifies","modality":"factual","temporality":"existing","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"implicit","scope":"Serienlösung|finale Produktion"} + ] + }, + { + "case_id": "g_no_decision", + "description": "Preference, alternative, and impersonal checking need without decision.", + "subject_id": "subject_g", + "subject": "Reale Recyclinganlage oder Technikum und verfügbarer Reinigungsansatz", + "evidence": [ + {"evidence_id":"e1","text":"Martin: Eine reale Recyclinganlage hätte das Risiko, dass wir kontaminiertes Material zurückbekommen."}, + {"evidence_id":"e2","text":"Martin: Ich würde nicht in eine reale Anlage gehen. Wenn überhaupt, können wir über ein Technikum reden."}, + {"evidence_id":"e3","text":"Tim: Man müsste zunächst prüfen, welcher Reinigungsansatz überhaupt verfügbar ist."} + ], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Eine reale Recyclinganlage birgt das Risiko kontaminierten Rückmaterials.","target":"discussion_subject","relation":"none","modality":"possible","temporality":"future","evaluation":"negative","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"none","scope":"reale Recyclinganlage|kontaminiertes Material"}, + {"observation_id":"obs_2","evidence_id":"e2","content":"Martin würde nicht in eine reale Anlage gehen.","target":"discussion_subject","relation":"opposes","modality":"suggested","temporality":"future","evaluation":"negative","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"Martins persönliche Präferenz|reale Anlage"}, + {"observation_id":"obs_3","evidence_id":"e2","content":"Ein Technikum bleibt als bedingte Möglichkeit im Gespräch.","target":"discussion_subject","relation":"none","modality":"possible","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"none","scope":"wenn überhaupt|Technikum"}, + {"observation_id":"obs_4","evidence_id":"e3","content":"Zunächst muss geprüft werden, welcher Reinigungsansatz verfügbar ist.","target":"discussion_subject","relation":"qualifies","modality":"impersonal_necessity","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"explicit","scope":"zunächst|verfügbarer Reinigungsansatz"} + ] + }, + { + "case_id": "h_resulting_action", + "description": "Interpersonal request followed by accepted responsibility.", + "subject_id": "subject_h", + "subject": "Prüfung der Messdaten bis Freitag", + "evidence": [ + {"evidence_id":"e1","text":"Antonius: Nina, übernimmst du die Prüfung der Messdaten bis Freitag?"}, + {"evidence_id":"e2","text":"Nina: Ja, ich übernehme die Prüfung bis Freitag."} + ], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Antonius bittet Nina um die Prüfung der Messdaten.","target":"discussion_subject","relation":"none","modality":"interpersonal_request","temporality":"future","evaluation":"none","agreement":"none","responsibility":"named","person":"Nina","uncertainty":"absent","clarification_need":"none","scope":"bis Freitag|Freitag"}, + {"observation_id":"obs_2","evidence_id":"e2","content":"Nina übernimmt die Prüfung.","target":"obs_1","relation":"supports","modality":"committed","temporality":"future","evaluation":"none","agreement":"accepted","responsibility":"accepted","person":"Nina","uncertainty":"absent","clarification_need":"none","scope":"bis Freitag|Freitag"} + ] + }, + { + "case_id": "i_outcome_and_unresolved", + "description": "Bounded production finding and unresolved publication information.", + "subject_id": "subject_i", + "subject": "Produktionsaufwand und Veröffentlichung von Energieaudit-Daten", + "evidence": [ + {"evidence_id":"e1","text":"Martin: An unserer Anlage gab es bei der reinen Produktion gegenüber dem Standardprodukt praktisch keine Änderung; wir waren nur fünf Grad kälter."}, + {"evidence_id":"e2","text":"Antonius: Dann können wir mindestens festhalten: Gegenüber Virgin Material ist bei der reinen Produktion kein zusätzlicher Aufwand notwendig. Davor entsteht natürlich Aufwand."}, + {"evidence_id":"e3","text":"Antonius: Welche Daten aus dem Energieaudit dürfen wir veröffentlichen?"}, + {"evidence_id":"e4","text":"Martin: Das ist weiterhin ungeklärt. Wir müssen die Freigabe noch klären."} + ], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Bei der reinen Produktion gab es praktisch keine Änderung gegenüber dem Standardprodukt.","target":"discussion_subject","relation":"none","modality":"factual","temporality":"completed","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"eigene Anlage, reine Produktion, Standardprodukt|reine Produktion"}, + {"observation_id":"obs_2","evidence_id":"e1","content":"Die Produktion erfolgte fünf Grad kälter.","target":"obs_1","relation":"qualifies","modality":"factual","temporality":"completed","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"fünf Grad kälter|5 Grad"}, + {"observation_id":"obs_3","evidence_id":"e2","content":"Gegenüber Virgin Material ist bei reiner Produktion kein zusätzlicher Aufwand notwendig.","target":"obs_1","relation":"supports","modality":"factual","temporality":"existing","evaluation":"none","agreement":"accepted","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"reine Produktion gegenüber Virgin Material|Virgin Material"}, + {"observation_id":"obs_4","evidence_id":"e2","content":"Vor der reinen Produktion entsteht Aufwand.","target":"obs_3","relation":"limits_scope","modality":"factual","temporality":"existing","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"vor der reinen Produktion|davor"}, + {"observation_id":"obs_5","evidence_id":"e3","content":"Es wird gefragt, welche Energieaudit-Daten veröffentlicht werden dürfen.","target":"discussion_subject","relation":"none","modality":"information_question","temporality":"unspecified","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"explicit","scope":"Veröffentlichung von Energieaudit-Daten|Energieaudit"}, + {"observation_id":"obs_6","evidence_id":"e4","content":"Die Veröffentlichungserlaubnis ist weiterhin ungeklärt.","target":"obs_5","relation":"supports","modality":"factual","temporality":"existing","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"explicit","scope":"Veröffentlichungserlaubnis|Freigabe"}, + {"observation_id":"obs_7","evidence_id":"e4","content":"Die Freigabe muss noch geklärt werden.","target":"obs_5","relation":"supports","modality":"impersonal_necessity","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"explicit","scope":"Freigabe zur Veröffentlichung|Freigabe"} + ] + } + ] +} diff --git a/tests/gold/evidence_observations_v2/cases.json b/tests/gold/evidence_observations_v2/cases.json new file mode 100644 index 0000000..4d5aa9b --- /dev/null +++ b/tests/gold/evidence_observations_v2/cases.json @@ -0,0 +1,99 @@ +{ + "cases": [ + { + "case_id": "a_idea_only", "description": "Possible geometry optimization without commitment.", + "subject_id": "subject_a", "subject": "Optimierung der Geometrie", + "evidence": [{"evidence_id": "e1", "text": "Martin: Die Geometrie kann man vielleicht noch optimieren. Dann würde man mal gucken, was herauskommt."}], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Die Geometrie kann vielleicht optimiert werden.","refers_to":null,"speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":true,"modality":"possible","temporality":"future","evaluation":"positive","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"none","qualifier":null,"limits_target":null}, + {"observation_id":"obs_2","evidence_id":"e1","content":"Danach könnte betrachtet werden, was herauskommt.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":true,"modality":"suggested","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"implicit","qualifier":"danach","limits_target":null} + ] + }, + { + "case_id": "b_multiple_options", "description": "Two alternatives for insufficient grid strength.", + "subject_id": "subject_b", "subject": "Umgang mit unzureichender Festigkeit des 40-40-Gitters", + "evidence": [{"evidence_id":"e1","text":"Martin: Die Festigkeit reicht für das 40-40-Gitter noch nicht aus."},{"evidence_id":"e2","text":"Martin: Man könnte mehr Masse für die gleiche Festigkeit einsetzen."},{"evidence_id":"e3","text":"Martin: Oder wir verkaufen es nicht als 40-40-Gitter, sondern machen ein 20-20 daraus. Das wären die zwei Ansätze."}], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Die Festigkeit des 40-40-Gitters reicht noch nicht aus.","refers_to":null,"speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"negative","affirmation":"absent","negation":"explicit","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"40-40-Gitter","limits_target":null}, + {"observation_id":"obs_2","evidence_id":"e2","content":"Mehr Masse könnte für die gleiche Festigkeit eingesetzt werden.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":true,"modality":"possible","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"mehr Masse für die gleiche Festigkeit","limits_target":null}, + {"observation_id":"obs_3","evidence_id":"e3","content":"Das Produkt könnte als 20-20 statt 40-40 ausgeführt werden.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"suggested","temporality":"future","evaluation":"none","affirmation":"absent","negation":"explicit","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"20-20 statt 40-40","limits_target":null}, + {"observation_id":"obs_4","evidence_id":"e3","content":"Die vorherigen Möglichkeiten sind die zwei Ansätze.","refers_to":null,"speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"zwei Ansätze","limits_target":null} + ] + }, + { + "case_id": "c_unaccepted_proposal", "description": "Suggested Textor contact without established work.", + "subject_id": "subject_c", "subject": "Erneute Kontaktaufnahme mit Dirk Textor zur Einschätzung", + "evidence": [{"evidence_id":"e1","text":"Tim: Ich würde vielleicht Dirk Textor noch einmal kontaktieren und fragen, wie er das einschätzt."},{"evidence_id":"e2","text":"Tim: Das kann man ja mit ihm einfach noch einmal rückkoppeln."}], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Tim erwägt, Dirk Textor erneut zu kontaktieren und nach seiner Einschätzung zu fragen.","refers_to":null,"speaker":"Tim","named_person":"Dirk Textor","addressee":null,"self_reference":true,"collective_we":false,"impersonal_person_reference":false,"modality":"suggested","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"none","qualifier":"erneut; Dirk Textors Einschätzung","limits_target":null}, + {"observation_id":"obs_2","evidence_id":"e2","content":"Eine erneute Rückkopplung mit Dirk Textor ist möglich.","refers_to":"obs_1","speaker":"Tim","named_person":"Dirk Textor","addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":true,"modality":"possible","temporality":"future","evaluation":"positive","affirmation":"explicit","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"noch einmal mit ihm","limits_target":null} + ] + }, + { + "case_id": "d_proposal_with_objection", "description": "Washing possibility and explicit energy disadvantage.", + "subject_id": "subject_d", "subject": "Waschen des Materials vor der weiteren Verarbeitung", + "evidence": [{"evidence_id":"e1","text":"Antonius: Man könnte das Material vor der weiteren Verarbeitung waschen."},{"evidence_id":"e2","text":"Martin: Ob sich das lohnt, weiß ich nicht. Waschen heißt nass machen und wieder trocknen; das ist ein wahnsinniger Energieaufwand."}], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Das Material könnte gewaschen werden.","refers_to":null,"speaker":"Antonius","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":true,"modality":"possible","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"vor der weiteren Verarbeitung","limits_target":null}, + {"observation_id":"obs_2","evidence_id":"e2","content":"Martin weiß nicht, ob sich das Waschen lohnt.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":true,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"explicit","determination_statement":"absent","uncertainty":"present","clarification_need":"none","qualifier":"ob es sich lohnt","limits_target":null}, + {"observation_id":"obs_3","evidence_id":"e2","content":"Waschen umfasst Nassmachen und erneutes Trocknen.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"nass machen und wieder trocknen","limits_target":null}, + {"observation_id":"obs_4","evidence_id":"e2","content":"Waschen und Trocknen verursachen einen sehr hohen Energieaufwand.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"negative","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"Waschen und Trocknen","limits_target":null} + ] + }, + { + "case_id": "e_rejected_alternative", "description": "Explicit negation followed by confirmation of that determination.", + "subject_id": "subject_e", "subject": "Zusammenarbeit mit Dr. Schlummer für Versuche", + "evidence": [{"evidence_id":"e1","text":"Antonius: Das Angebot von Dr. Schlummer für die Versuche kostet 30.000 Euro."},{"evidence_id":"e2","text":"Tim: Dann haben wir gesagt: Nein, die Zusammenarbeit mit Dr. Schlummer machen wir nicht."},{"evidence_id":"e3","text":"Antonius: Ja, das ist entschieden."}], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Das Angebot von Dr. Schlummer kostet 30.000 Euro.","refers_to":null,"speaker":"Antonius","named_person":"Dr. Schlummer","addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"für die Versuche; 30.000 Euro","limits_target":null}, + {"observation_id":"obs_2","evidence_id":"e2","content":"Die Zusammenarbeit mit Dr. Schlummer wird nicht durchgeführt.","refers_to":null,"speaker":"Tim","named_person":"Dr. Schlummer","addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"committed","temporality":"future","evaluation":"none","affirmation":"absent","negation":"explicit","determination_statement":"present","uncertainty":"absent","clarification_need":"none","qualifier":"Zusammenarbeit für die Versuche","limits_target":null}, + {"observation_id":"obs_3","evidence_id":"e3","content":"Die vorherige Festlegung ist entschieden.","refers_to":"obs_2","speaker":"Antonius","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"completed","evaluation":"none","affirmation":"explicit","negation":"absent","determination_statement":"present","uncertainty":"absent","clarification_need":"none","qualifier":null,"limits_target":null} + ] + }, + { + "case_id": "f_trial_only_acceptance", "description": "Affirmed commitment limited to a 20-metre trial.", + "subject_id": "subject_f", "subject": "20-Prozent-Variante im Versuch am kleinen Extruder", + "evidence": [{"evidence_id":"e1","text":"Martin: Wir könnten die 20-Prozent-Variante am kleinen Extruder nachstellen."},{"evidence_id":"e2","text":"Tim: Ja, wir testen 20 Meter dieser Variante beim nächsten Versuch."},{"evidence_id":"e3","text":"Tim: Das ist nur ein Versuch; damit ist die Variante noch nicht als Serienlösung festgelegt."}], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Die 20-Prozent-Variante könnte am kleinen Extruder nachgestellt werden.","refers_to":null,"speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"possible","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"am kleinen Extruder","limits_target":null}, + {"observation_id":"obs_2","evidence_id":"e2","content":"20 Meter der Variante werden beim nächsten Versuch getestet.","refers_to":"obs_1","speaker":"Tim","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"committed","temporality":"future","evaluation":"none","affirmation":"explicit","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"20 Meter beim nächsten Versuch","limits_target":null}, + {"observation_id":"obs_3","evidence_id":"e3","content":"Die Zusage gilt nur für einen Versuch.","refers_to":"obs_2","speaker":"Tim","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"nur ein Versuch","limits_target":"obs_2"}, + {"observation_id":"obs_4","evidence_id":"e3","content":"Die Variante ist noch nicht als Serienlösung festgelegt.","refers_to":"obs_2","speaker":"Tim","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"explicit","determination_statement":"present","uncertainty":"present","clarification_need":"implicit","qualifier":"als Serienlösung","limits_target":null} + ] + }, + { + "case_id": "g_no_decision", "description": "Preference, alternative, and impersonal checking need without decision.", + "subject_id": "subject_g", "subject": "Reale Recyclinganlage oder Technikum und verfügbarer Reinigungsansatz", + "evidence": [{"evidence_id":"e1","text":"Martin: Eine reale Recyclinganlage hätte das Risiko, dass wir kontaminiertes Material zurückbekommen."},{"evidence_id":"e2","text":"Martin: Ich würde nicht in eine reale Anlage gehen. Wenn überhaupt, können wir über ein Technikum reden."},{"evidence_id":"e3","text":"Tim: Man müsste zunächst prüfen, welcher Reinigungsansatz überhaupt verfügbar ist."}], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Eine reale Recyclinganlage birgt das Risiko kontaminierten Rückmaterials.","refers_to":null,"speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"possible","temporality":"future","evaluation":"negative","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"none","qualifier":"reale Recyclinganlage; kontaminiertes Material","limits_target":null}, + {"observation_id":"obs_2","evidence_id":"e2","content":"Martin würde persönlich nicht in eine reale Anlage gehen.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":true,"collective_we":false,"impersonal_person_reference":false,"modality":"suggested","temporality":"future","evaluation":"negative","affirmation":"absent","negation":"explicit","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"reale Anlage","limits_target":null}, + {"observation_id":"obs_3","evidence_id":"e2","content":"Ein Technikum bleibt als bedingte Möglichkeit im Gespräch.","refers_to":null,"speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"possible","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"none","qualifier":"wenn überhaupt; Technikum","limits_target":null}, + {"observation_id":"obs_4","evidence_id":"e3","content":"Zunächst muss geprüft werden, welcher Reinigungsansatz verfügbar ist.","refers_to":null,"speaker":"Tim","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":true,"modality":"impersonal_necessity","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"explicit","qualifier":"zunächst; verfügbarer Reinigungsansatz","limits_target":null} + ] + }, + { + "case_id": "h_resulting_action", "description": "Interpersonal request followed by explicit personal acceptance.", + "subject_id": "subject_h", "subject": "Prüfung der Messdaten bis Freitag", + "evidence": [{"evidence_id":"e1","text":"Antonius: Nina, übernimmst du die Prüfung der Messdaten bis Freitag?"},{"evidence_id":"e2","text":"Nina: Ja, ich übernehme die Prüfung bis Freitag."}], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Antonius richtet an Nina die Bitte, die Messdaten zu prüfen.","refers_to":null,"speaker":"Antonius","named_person":"Nina","addressee":"Nina","self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"interpersonal_request","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"bis Freitag","limits_target":null}, + {"observation_id":"obs_2","evidence_id":"e2","content":"Nina sagt zu, die Prüfung zu übernehmen.","refers_to":"obs_1","speaker":"Nina","named_person":null,"addressee":null,"self_reference":true,"collective_we":false,"impersonal_person_reference":false,"modality":"committed","temporality":"future","evaluation":"none","affirmation":"explicit","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"bis Freitag","limits_target":null} + ] + }, + { + "case_id": "i_outcome_and_unresolved", "description": "Bounded production finding and unresolved publication information.", + "subject_id": "subject_i", "subject": "Produktionsaufwand und Veröffentlichung von Energieaudit-Daten", + "evidence": [{"evidence_id":"e1","text":"Martin: An unserer Anlage gab es bei der reinen Produktion gegenüber dem Standardprodukt praktisch keine Änderung; wir waren nur fünf Grad kälter."},{"evidence_id":"e2","text":"Antonius: Dann können wir mindestens festhalten: Gegenüber Virgin Material ist bei der reinen Produktion kein zusätzlicher Aufwand notwendig. Davor entsteht natürlich Aufwand."},{"evidence_id":"e3","text":"Antonius: Welche Daten aus dem Energieaudit dürfen wir veröffentlichen?"},{"evidence_id":"e4","text":"Martin: Das ist weiterhin ungeklärt. Wir müssen die Freigabe noch klären."}], + "expected_observations": [ + {"observation_id":"obs_1","evidence_id":"e1","content":"Bei der reinen Produktion gab es praktisch keine Änderung gegenüber dem Standardprodukt.","refers_to":null,"speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"completed","evaluation":"none","affirmation":"absent","negation":"explicit","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"an unserer Anlage; reine Produktion; gegenüber dem Standardprodukt","limits_target":null}, + {"observation_id":"obs_2","evidence_id":"e1","content":"Die Produktion erfolgte fünf Grad kälter.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"factual","temporality":"completed","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"fünf Grad kälter","limits_target":null}, + {"observation_id":"obs_3","evidence_id":"e2","content":"Gegenüber Virgin Material ist bei reiner Produktion kein zusätzlicher Aufwand notwendig.","refers_to":"obs_1","speaker":"Antonius","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"explicit","determination_statement":"present","uncertainty":"absent","clarification_need":"none","qualifier":"bei reiner Produktion; gegenüber Virgin Material","limits_target":null}, + {"observation_id":"obs_4","evidence_id":"e2","content":"Vor der reinen Produktion entsteht Aufwand.","refers_to":"obs_3","speaker":"Antonius","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"davor","limits_target":"obs_3"}, + {"observation_id":"obs_5","evidence_id":"e3","content":"Es wird gefragt, welche Energieaudit-Daten veröffentlicht werden dürfen.","refers_to":null,"speaker":"Antonius","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"information_question","temporality":"unspecified","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"explicit","qualifier":"Veröffentlichung von Energieaudit-Daten","limits_target":null}, + {"observation_id":"obs_6","evidence_id":"e4","content":"Die Veröffentlichungserlaubnis ist weiterhin ungeklärt.","refers_to":"obs_5","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"explicit","qualifier":"weiterhin","limits_target":null}, + {"observation_id":"obs_7","evidence_id":"e4","content":"Die Freigabe muss noch geklärt werden.","refers_to":"obs_5","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"impersonal_necessity","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"explicit","qualifier":"noch; Freigabe zur Veröffentlichung","limits_target":null} + ] + } + ] +} diff --git a/tests/gold/evidence_observations_v3/cases.json b/tests/gold/evidence_observations_v3/cases.json new file mode 100644 index 0000000..aa1fec9 --- /dev/null +++ b/tests/gold/evidence_observations_v3/cases.json @@ -0,0 +1,49 @@ +{ + "cases": [ + { + "case_id":"a_idea_only","description":"Possible geometry optimization without commitment.","subject_id":"subject_a","subject":"Optimierung der Geometrie", + "evidence":[{"evidence_id":"e1","text":"Martin: Die Geometrie kann man vielleicht noch optimieren. Dann würde man mal gucken, was herauskommt."}], + "semantic_requirements":["Geometry optimization remains possible and tentative.","Subsequent checking remains conditional and tentative.","The then/sequential dependency survives.","No commitment or owner is introduced."] + }, + { + "case_id":"b_multiple_options","description":"Two alternatives for insufficient grid strength.","subject_id":"subject_b","subject":"Umgang mit unzureichender Festigkeit des 40-40-Gitters", + "evidence":[{"evidence_id":"e1","text":"Martin: Die Festigkeit reicht für das 40-40-Gitter noch nicht aus."},{"evidence_id":"e2","text":"Martin: Man könnte mehr Masse für die gleiche Festigkeit einsetzen."},{"evidence_id":"e3","text":"Martin: Oder wir verkaufen es nicht als 40-40-Gitter, sondern machen ein 20-20 daraus. Das wären die zwei Ansätze."}], + "semantic_requirements":["Insufficient 40-40 strength survives.","Additional mass remains one alternative.","20-20 remains another alternative.","Both remain alternatives and neither is selected."] + }, + { + "case_id":"c_unaccepted_proposal","description":"Suggested Textor contact without established work.","subject_id":"subject_c","subject":"Erneute Kontaktaufnahme mit Dirk Textor zur Einschätzung", + "evidence":[{"evidence_id":"e1","text":"Tim: Ich würde vielleicht Dirk Textor noch einmal kontaktieren und fragen, wie er das einschätzt."},{"evidence_id":"e2","text":"Tim: Das kann man ja mit ihm einfach noch einmal rückkoppeln."}], + "semantic_requirements":["Contacting Dirk Textor remains Tim's tentative personal suggestion.","The follow-up remains possible and relates to that contact.","No established work or responsibility is introduced."] + }, + { + "case_id":"d_proposal_with_objection","description":"Washing possibility and explicit energy disadvantage.","subject_id":"subject_d","subject":"Waschen des Materials vor der weiteren Verarbeitung", + "evidence":[{"evidence_id":"e1","text":"Antonius: Man könnte das Material vor der weiteren Verarbeitung waschen."},{"evidence_id":"e2","text":"Martin: Ob sich das lohnt, weiß ich nicht. Waschen heißt nass machen und wieder trocknen; das ist ein wahnsinniger Energieaufwand."}], + "semantic_requirements":["Washing before further processing remains possible.","Martin's uncertainty whether washing is worthwhile survives.","The washing and drying process survives.","The high energy consequence survives.","No unresolved task is invented."] + }, + { + "case_id":"e_rejected_alternative","description":"Explicit rejection followed by confirmation of that determination.","subject_id":"subject_e","subject":"Zusammenarbeit mit Dr. Schlummer für Versuche", + "evidence":[{"evidence_id":"e1","text":"Antonius: Das Angebot von Dr. Schlummer für die Versuche kostet 30.000 Euro."},{"evidence_id":"e2","text":"Tim: Dann haben wir gesagt: Nein, die Zusammenarbeit mit Dr. Schlummer machen wir nicht."},{"evidence_id":"e3","text":"Antonius: Ja, das ist entschieden."}], + "semantic_requirements":["The offer cost survives without inferred evaluation.","Collaboration is explicitly not to be pursued.","The explicit rejection survives.","The later statement confirms that the preceding determination has been made."] + }, + { + "case_id":"f_trial_only_acceptance","description":"Collective commitment limited to a 20-metre trial.","subject_id":"subject_f","subject":"20-Prozent-Variante im Versuch am kleinen Extruder", + "evidence":[{"evidence_id":"e1","text":"Martin: Wir könnten die 20-Prozent-Variante am kleinen Extruder nachstellen."},{"evidence_id":"e2","text":"Tim: Ja, wir testen 20 Meter dieser Variante beim nächsten Versuch."},{"evidence_id":"e3","text":"Tim: Das ist nur ein Versuch; damit ist die Variante noch nicht als Serienlösung festgelegt."}], + "semantic_requirements":["The 20-percent variant at the small extruder remains initially possible.","The later statement collectively commits to a test.","Twenty metres and next-trial timing survive.","The test remains limited to a trial.","Series adoption remains explicitly not yet established.","No individual owner is invented."] + }, + { + "case_id":"g_no_decision","description":"Preference, conditional alternative, and impersonal checking need.","subject_id":"subject_g","subject":"Reale Recyclinganlage oder Technikum und verfügbarer Reinigungsansatz", + "evidence":[{"evidence_id":"e1","text":"Martin: Eine reale Recyclinganlage hätte das Risiko, dass wir kontaminiertes Material zurückbekommen."},{"evidence_id":"e2","text":"Martin: Ich würde nicht in eine reale Anlage gehen. Wenn überhaupt, können wir über ein Technikum reden."},{"evidence_id":"e3","text":"Tim: Man müsste zunächst prüfen, welcher Reinigungsansatz überhaupt verfügbar ist."}], + "semantic_requirements":["Contamination remains a risk rather than a fact.","Martin's negative stance remains personal.","The Technikum remains conditional and if-at-all survives.","Cleaning-method availability still needs to be checked.","The need remains impersonal.","No group decision or owner is invented."] + }, + { + "case_id":"h_resulting_action","description":"Interpersonal request followed by explicit personal acceptance.","subject_id":"subject_h","subject":"Prüfung der Messdaten bis Freitag", + "evidence":[{"evidence_id":"e1","text":"Antonius: Nina, übernimmst du die Prüfung der Messdaten bis Freitag?"},{"evidence_id":"e2","text":"Nina: Ja, ich übernehme die Prüfung bis Freitag."}], + "semantic_requirements":["Antonius requests measurement-data review from Nina.","The Friday deadline survives.","Nina explicitly accepts the preceding request.","Nina's response expresses future personal commitment.","No responsibility field or unsupported inference is introduced."] + }, + { + "case_id":"i_outcome_and_unresolved","description":"Bounded production finding and unresolved publication information.","subject_id":"subject_i","subject":"Produktionsaufwand und Veröffentlichung von Energieaudit-Daten", + "evidence":[{"evidence_id":"e1","text":"Martin: An unserer Anlage gab es bei der reinen Produktion gegenüber dem Standardprodukt praktisch keine Änderung; wir waren nur fünf Grad kälter."},{"evidence_id":"e2","text":"Antonius: Dann können wir mindestens festhalten: Gegenüber Virgin Material ist bei der reinen Produktion kein zusätzlicher Aufwand notwendig. Davor entsteht natürlich Aufwand."},{"evidence_id":"e3","text":"Antonius: Welche Daten aus dem Energieaudit dürfen wir veröffentlichen?"},{"evidence_id":"e4","text":"Martin: Das ist weiterhin ungeklärt. Wir müssen die Freigabe noch klären."}], + "semantic_requirements":["The pure-production finding remains bounded to the local plant and standard-product comparison.","The five-degree difference survives.","No-extra-effort remains bounded to pure production compared with Virgin material.","Upstream effort before that production stage survives.","The publication purpose of the energy-audit question remains explicit.","Publication permission remains unresolved and clarification remains necessary.","No assigned work is invented."] + } + ] +} diff --git a/tests/gold/semantic_synthesis_isolation/README.md b/tests/gold/semantic_synthesis_isolation/README.md new file mode 100644 index 0000000..3093cce --- /dev/null +++ b/tests/gold/semantic_synthesis_isolation/README.md @@ -0,0 +1,11 @@ +# semantic_synthesis_isolation + +Isolation Gold set derived from the existing Topic Reconstruction V2 A-I +cases. Every case supplies one manually fixed Discussion Subject and the +complete original evidence bundle. The model performs Semantic Synthesis only; +subject detection, subject grouping, and evidence assignment are outside the +experiment. + +Expected criteria evaluate semantic event distinctions, outcomes and scope, +actions, unresolved issues, and supporting evidence. They do not evaluate +subject discovery or exact wording. diff --git a/tests/gold/semantic_synthesis_isolation/cases.json b/tests/gold/semantic_synthesis_isolation/cases.json new file mode 100644 index 0000000..67104fe --- /dev/null +++ b/tests/gold/semantic_synthesis_isolation/cases.json @@ -0,0 +1,227 @@ +{ + "cases": [ + { + "case_id": "a_idea_only", + "description": "Idea mentioned without stronger commitment.", + "subject_id": "subject_a", + "subject": "Optimierung der Geometrie", + "evidence": [ + {"evidence_id": "e1", "text": "Martin: Die Geometrie kann man vielleicht noch optimieren. Dann würde man mal gucken, was herauskommt."} + ], + "allowed_responsible": [], + "expected": { + "event_type_minimums": {"idea": 1}, + "allowed_event_types": ["idea"], + "event_evidence_ids": ["e1"], + "outcome": {"required": false}, + "actions": {"count": 0}, + "unresolved_issues": {"count": 0} + } + }, + { + "case_id": "b_multiple_options", + "description": "Two alternatives for insufficient 40-40 grid strength.", + "subject_id": "subject_b", + "subject": "Umgang mit unzureichender Festigkeit des 40-40-Gitters", + "evidence": [ + {"evidence_id": "e1", "text": "Martin: Die Festigkeit reicht für das 40-40-Gitter noch nicht aus."}, + {"evidence_id": "e2", "text": "Martin: Man könnte mehr Masse für die gleiche Festigkeit einsetzen."}, + {"evidence_id": "e3", "text": "Martin: Oder wir verkaufen es nicht als 40-40-Gitter, sondern machen ein 20-20 daraus. Das wären die zwei Ansätze."} + ], + "allowed_responsible": [], + "expected": { + "event_type_minimums": {"option": 2}, + "allowed_event_types": ["technical_finding", "fact", "option"], + "event_evidence_ids": ["e1", "e2", "e3"], + "outcome": {"required": false}, + "actions": {"count": 0}, + "unresolved_issues": {"count": 0} + } + }, + { + "case_id": "c_unaccepted_proposal", + "description": "Possible Textor contact remains a proposal only.", + "subject_id": "subject_c", + "subject": "Erneute Kontaktaufnahme mit Dirk Textor zur Einschätzung", + "evidence": [ + {"evidence_id": "e1", "text": "Tim: Ich würde vielleicht Dirk Textor noch einmal kontaktieren und fragen, wie er das einschätzt."}, + {"evidence_id": "e2", "text": "Tim: Das kann man ja mit ihm einfach noch einmal rückkoppeln."} + ], + "allowed_responsible": [], + "expected": { + "event_type_minimums": {"proposal": 1}, + "allowed_event_types": ["proposal"], + "event_evidence_ids": ["e1", "e2"], + "outcome": {"required": false}, + "actions": {"count": 0}, + "unresolved_issues": {"count": 0} + } + }, + { + "case_id": "d_proposal_with_objection", + "description": "Washing proposal with energy objection but no unresolved issue.", + "subject_id": "subject_d", + "subject": "Waschen des Materials vor der weiteren Verarbeitung", + "evidence": [ + {"evidence_id": "e1", "text": "Antonius: Man könnte das Material vor der weiteren Verarbeitung waschen."}, + {"evidence_id": "e2", "text": "Martin: Ob sich das lohnt, weiß ich nicht. Waschen heißt nass machen und wieder trocknen; das ist ein wahnsinniger Energieaufwand."} + ], + "allowed_responsible": [], + "expected": { + "event_type_minimums": {"proposal": 1, "objection": 1}, + "allowed_event_types": ["proposal", "objection"], + "event_evidence_ids": ["e1", "e2"], + "outcome": {"required": false}, + "actions": {"count": 0}, + "unresolved_issues": {"count": 0} + } + }, + { + "case_id": "e_rejected_alternative", + "description": "Explicit rejection of Schlummer collaboration.", + "subject_id": "subject_e", + "subject": "Zusammenarbeit mit Dr. Schlummer für Versuche", + "evidence": [ + {"evidence_id": "e1", "text": "Antonius: Das Angebot von Dr. Schlummer für die Versuche kostet 30.000 Euro."}, + {"evidence_id": "e2", "text": "Tim: Dann haben wir gesagt: Nein, die Zusammenarbeit mit Dr. Schlummer machen wir nicht."}, + {"evidence_id": "e3", "text": "Antonius: Ja, das ist entschieden."} + ], + "allowed_responsible": [], + "expected": { + "event_type_minimums": {"fact": 1, "rejection": 1}, + "allowed_event_types": ["fact", "rejection", "clarification"], + "event_evidence_ids": ["e1", "e2", "e3"], + "outcome": { + "required": true, + "statuses": ["rejected"], + "terms": ["nicht", "abgelehnt", "keine"], + "scope_terms": ["zusammenarbeit", "versuch", "schlummer"], + "evidence_ids": ["e2", "e3"] + }, + "actions": {"count": 0}, + "unresolved_issues": {"count": 0} + } + }, + { + "case_id": "f_trial_only_acceptance", + "description": "Acceptance limited to a 20-metre trial.", + "subject_id": "subject_f", + "subject": "20-Prozent-Variante im Versuch am kleinen Extruder", + "evidence": [ + {"evidence_id": "e1", "text": "Martin: Wir könnten die 20-Prozent-Variante am kleinen Extruder nachstellen."}, + {"evidence_id": "e2", "text": "Tim: Ja, wir testen 20 Meter dieser Variante beim nächsten Versuch."}, + {"evidence_id": "e3", "text": "Tim: Das ist nur ein Versuch; damit ist die Variante noch nicht als Serienlösung festgelegt."} + ], + "allowed_responsible": [], + "expected": { + "event_type_minimums": {"proposal": 1, "scoped_acceptance": 1, "clarification": 1}, + "allowed_event_types": ["proposal", "scoped_acceptance", "clarification"], + "event_evidence_ids": ["e1", "e2", "e3"], + "outcome": { + "required": true, + "statuses": ["scoped_acceptance"], + "terms": ["test", "versuch"], + "scope_terms": ["20 meter", "20 m", "nur", "begrenzt"], + "evidence_ids": ["e2", "e3"] + }, + "actions": { + "count": 1, + "terms": ["test", "versuch", "20 meter"], + "evidence_ids": ["e2"], + "due_terms": ["nächsten versuch", "next trial"] + }, + "unresolved_issues": { + "count": 1, + "terms": ["serienlösung", "final", "serie", "festgelegt"], + "evidence_ids": ["e3"] + } + } + }, + { + "case_id": "g_no_decision", + "description": "Plant versus Technikum discussion ending without a decision.", + "subject_id": "subject_g", + "subject": "Reale Recyclinganlage oder Technikum und verfügbarer Reinigungsansatz", + "evidence": [ + {"evidence_id": "e1", "text": "Martin: Eine reale Recyclinganlage hätte das Risiko, dass wir kontaminiertes Material zurückbekommen."}, + {"evidence_id": "e2", "text": "Martin: Ich würde nicht in eine reale Anlage gehen. Wenn überhaupt, können wir über ein Technikum reden."}, + {"evidence_id": "e3", "text": "Tim: Man müsste zunächst prüfen, welcher Reinigungsansatz überhaupt verfügbar ist."} + ], + "allowed_responsible": [], + "expected": { + "event_type_minimums": {"objection": 1, "option": 1}, + "allowed_event_types": ["objection", "option", "proposal", "clarification"], + "event_evidence_ids": ["e1", "e2", "e3"], + "outcome": {"required": false}, + "actions": {"count": 0}, + "unresolved_issues": { + "count": 1, + "terms": ["reinigungsansatz", "verfügbar", "prüfen", "reinigung"], + "evidence_ids": ["e3"] + } + } + }, + { + "case_id": "h_resulting_action", + "description": "Explicitly accepted action with owner and deadline.", + "subject_id": "subject_h", + "subject": "Prüfung der Messdaten bis Freitag", + "evidence": [ + {"evidence_id": "e1", "text": "Antonius: Nina, übernimmst du die Prüfung der Messdaten bis Freitag?"}, + {"evidence_id": "e2", "text": "Nina: Ja, ich übernehme die Prüfung bis Freitag."} + ], + "allowed_responsible": ["Nina"], + "expected": { + "event_type_minimums": {}, + "allowed_event_types": ["proposal", "clarification", "scoped_acceptance"], + "event_evidence_ids": [], + "outcome": { + "required": true, + "statuses": ["established"], + "terms": ["übernimmt", "prüf", "accepted", "review", "assigned"], + "scope_terms": ["messdaten", "prüfung", "measurement", "review"], + "evidence_ids": ["e2"] + }, + "actions": { + "count": 1, + "terms": ["messdaten", "prüf", "measurement", "review"], + "responsible": "Nina", + "due_terms": ["freitag", "friday"], + "evidence_ids": ["e2"] + }, + "unresolved_issues": {"count": 0} + } + }, + { + "case_id": "i_outcome_and_unresolved", + "description": "Bounded production outcome and unresolved publication question.", + "subject_id": "subject_i", + "subject": "Produktionsaufwand und Veröffentlichung von Energieaudit-Daten", + "evidence": [ + {"evidence_id": "e1", "text": "Martin: An unserer Anlage gab es bei der reinen Produktion gegenüber dem Standardprodukt praktisch keine Änderung; wir waren nur fünf Grad kälter."}, + {"evidence_id": "e2", "text": "Antonius: Dann können wir mindestens festhalten: Gegenüber Virgin Material ist bei der reinen Produktion kein zusätzlicher Aufwand notwendig. Davor entsteht natürlich Aufwand."}, + {"evidence_id": "e3", "text": "Antonius: Welche Daten aus dem Energieaudit dürfen wir veröffentlichen?"}, + {"evidence_id": "e4", "text": "Martin: Das ist weiterhin ungeklärt. Wir müssen die Freigabe noch klären."} + ], + "allowed_responsible": [], + "expected": { + "event_type_minimums": {"fact": 1}, + "allowed_event_types": ["technical_finding", "fact", "clarification"], + "event_evidence_ids": ["e1", "e2"], + "outcome": { + "required": true, + "statuses": ["established"], + "terms": ["kein zusätzlicher", "keine zusätzliche", "unverändert"], + "scope_terms": ["reine produktion", "produktion", "virgin"], + "evidence_ids": ["e1", "e2"] + }, + "actions": {"count": 0}, + "unresolved_issues": { + "count": 1, + "terms": ["veröffentlich", "freigabe", "energieaudit", "daten"], + "evidence_ids": ["e3", "e4"] + } + } + } + ] +} diff --git a/tests/gold/topic_reconstruction_v2/README.md b/tests/gold/topic_reconstruction_v2/README.md new file mode 100644 index 0000000..ad35eb5 --- /dev/null +++ b/tests/gold/topic_reconstruction_v2/README.md @@ -0,0 +1,20 @@ +# topic_reconstruction_v2 + +Focused experimental Gold material derived from BUG-015 and the Progeo +discussion. These cases evaluate topic-oriented reconstruction rather than +exact protocol wording or flat category extraction. + +The nine cases cover: + +- an idea mentioned without further development; +- multiple alternatives; +- an unaccepted proposal; +- a proposal with an objection; +- an explicitly rejected alternative; +- acceptance limited to a bounded trial; +- discussion ending without a decision; +- a resulting Action Item; +- an outcome accompanied by an unresolved issue. + +Evidence units carry stable local IDs. Expected criteria describe semantic +features and prohibited promotions rather than exact generated sentences. diff --git a/tests/gold/topic_reconstruction_v2/cases.json b/tests/gold/topic_reconstruction_v2/cases.json new file mode 100644 index 0000000..745b33d --- /dev/null +++ b/tests/gold/topic_reconstruction_v2/cases.json @@ -0,0 +1,258 @@ +{ + "cases": [ + { + "case_id": "a_idea_only", + "description": "A geometry optimization idea is mentioned but not developed.", + "evidence_units": [ + { + "evidence_id": "e1", + "text": "Martin: Die Geometrie kann man vielleicht noch optimieren. Dann würde man mal gucken, was herauskommt." + } + ], + "expected": { + "subject_count": 1, + "subject_terms": ["geometr"], + "required_event_types": ["introduced_idea"], + "outcome": {"required": false}, + "actions": {"minimum": 0}, + "unresolved": {"minimum": 0} + } + }, + { + "case_id": "b_multiple_options", + "description": "Two alternatives for compensating insufficient specimen strength are discussed.", + "evidence_units": [ + { + "evidence_id": "e1", + "text": "Martin: Die Festigkeit reicht für das 40-40-Gitter noch nicht aus." + }, + { + "evidence_id": "e2", + "text": "Martin: Man könnte mehr Masse für die gleiche Festigkeit einsetzen." + }, + { + "evidence_id": "e3", + "text": "Martin: Oder wir verkaufen es nicht als 40-40-Gitter, sondern machen ein 20-20 daraus. Das wären die zwei Ansätze." + } + ], + "expected": { + "subject_count": 1, + "subject_terms": ["festigkeit", "gitter", "geometr"], + "required_event_types": ["considered_option"], + "outcome": {"required": false}, + "actions": {"minimum": 0}, + "unresolved": {"minimum": 0} + } + }, + { + "case_id": "c_unaccepted_proposal", + "description": "Contacting Dirk Textor is proposed but not accepted as work.", + "evidence_units": [ + { + "evidence_id": "e1", + "text": "Tim: Ich würde vielleicht Dirk Textor noch einmal kontaktieren und fragen, wie er das einschätzt." + }, + { + "evidence_id": "e2", + "text": "Tim: Das kann man ja mit ihm einfach noch einmal rückkoppeln." + } + ], + "expected": { + "subject_count": 1, + "subject_terms": ["textor", "einschätzung", "kontakt"], + "required_event_types": ["proposal"], + "outcome": {"required": false}, + "actions": {"minimum": 0}, + "unresolved": {"minimum": 0} + } + }, + { + "case_id": "d_proposal_with_objection", + "description": "Washing is considered and an energy-cost objection is raised.", + "evidence_units": [ + { + "evidence_id": "e1", + "text": "Antonius: Man könnte das Material vor der weiteren Verarbeitung waschen." + }, + { + "evidence_id": "e2", + "text": "Martin: Ob sich das lohnt, weiß ich nicht. Waschen heißt nass machen und wieder trocknen; das ist ein wahnsinniger Energieaufwand." + } + ], + "expected": { + "subject_count": 1, + "subject_terms": ["wasch", "reinig"], + "required_event_types": ["proposal", "objection"], + "outcome": {"required": false}, + "actions": {"minimum": 0}, + "unresolved": {"minimum": 0} + } + }, + { + "case_id": "e_rejected_alternative", + "description": "The collaboration with Dr. Schlummer is explicitly rejected after its cost is discussed.", + "evidence_units": [ + { + "evidence_id": "e1", + "text": "Antonius: Das Angebot von Dr. Schlummer für die Versuche kostet 30.000 Euro." + }, + { + "evidence_id": "e2", + "text": "Tim: Dann haben wir gesagt: Nein, die Zusammenarbeit mit Dr. Schlummer machen wir nicht." + }, + { + "evidence_id": "e3", + "text": "Antonius: Ja, das ist entschieden." + } + ], + "expected": { + "subject_count": 1, + "subject_terms": ["schlummer", "zusammenarbeit"], + "required_event_types": ["fact"], + "outcome": { + "required": true, + "terms": ["nicht", "abgelehnt", "keine"], + "scope_terms": ["zusammenarbeit", "versuch"], + "certainties": ["rejected", "established"] + }, + "actions": {"minimum": 0}, + "unresolved": {"minimum": 0} + } + }, + { + "case_id": "f_trial_only_acceptance", + "description": "A 20 percent variant is accepted only for a bounded trial, not as the final production solution.", + "evidence_units": [ + { + "evidence_id": "e1", + "text": "Martin: Wir könnten die 20-Prozent-Variante am kleinen Extruder nachstellen." + }, + { + "evidence_id": "e2", + "text": "Tim: Ja, wir testen 20 Meter dieser Variante beim nächsten Versuch." + }, + { + "evidence_id": "e3", + "text": "Tim: Das ist nur ein Versuch; damit ist die Variante noch nicht als Serienlösung festgelegt." + } + ], + "expected": { + "subject_count": 1, + "subject_terms": ["20-prozent", "variante", "extruder"], + "required_event_types": ["proposal", "clarification"], + "outcome": { + "required": true, + "terms": ["test", "versuch"], + "scope_terms": ["20 meter", "20 m", "nur", "begrenzt"], + "certainties": ["established", "conditional"] + }, + "actions": { + "minimum": 1, + "terms": ["test", "versuch", "20 meters", "20 meter"] + }, + "unresolved": { + "minimum": 1, + "terms": ["final", "series", "serie", "adopt", "festgelegt"] + } + } + }, + { + "case_id": "g_no_decision", + "description": "Real recycling plant and Technikum alternatives are discussed without a group decision.", + "evidence_units": [ + { + "evidence_id": "e1", + "text": "Martin: Eine reale Recyclinganlage hätte das Risiko, dass wir kontaminiertes Material zurückbekommen." + }, + { + "evidence_id": "e2", + "text": "Martin: Ich würde nicht in eine reale Anlage gehen. Wenn überhaupt, können wir über ein Technikum reden." + }, + { + "evidence_id": "e3", + "text": "Tim: Man müsste zunächst prüfen, welcher Reinigungsansatz überhaupt verfügbar ist." + } + ], + "expected": { + "subject_count": 1, + "subject_terms": ["technikum", "reinig", "anlage"], + "required_event_types": ["considered_option", "objection"], + "outcome": {"required": false}, + "actions": {"minimum": 0}, + "unresolved": { + "minimum": 1, + "terms": ["reinigungsansatz", "verfügbar", "anlage", "prüfen"] + } + } + }, + { + "case_id": "h_resulting_action", + "description": "The discussion establishes an accepted review action with owner and deadline.", + "evidence_units": [ + { + "evidence_id": "e1", + "text": "Antonius: Nina, übernimmst du die Prüfung der Messdaten bis Freitag?" + }, + { + "evidence_id": "e2", + "text": "Nina: Ja, ich übernehme die Prüfung bis Freitag." + } + ], + "expected": { + "subject_count": 1, + "subject_terms": ["messdaten", "prüfung", "prüfen", "verification", "data"], + "required_event_types": [], + "outcome": { + "required": true, + "terms": ["agrees", "übernimmt", "accepted", "verify"], + "scope_terms": ["measurement", "messdaten", "verification"], + "certainties": ["established"] + }, + "actions": { + "minimum": 1, + "terms": ["messdaten", "prüf", "verify", "measurement"], + "responsible": "Nina" + }, + "unresolved": {"minimum": 0} + } + }, + { + "case_id": "i_outcome_and_unresolved", + "description": "The production-energy discussion establishes one bounded finding while publication remains unresolved.", + "evidence_units": [ + { + "evidence_id": "e1", + "text": "Martin: An unserer Anlage gab es bei der reinen Produktion gegenüber dem Standardprodukt praktisch keine Änderung; wir waren nur fünf Grad kälter." + }, + { + "evidence_id": "e2", + "text": "Antonius: Dann können wir mindestens festhalten: Gegenüber Virgin Material ist bei der reinen Produktion kein zusätzlicher Aufwand notwendig. Davor entsteht natürlich Aufwand." + }, + { + "evidence_id": "e3", + "text": "Antonius: Welche Daten aus dem Energieaudit dürfen wir veröffentlichen?" + }, + { + "evidence_id": "e4", + "text": "Martin: Das ist weiterhin ungeklärt. Wir müssen die Freigabe noch klären." + } + ], + "expected": { + "subject_count": 1, + "subject_terms": ["energie", "aufwand", "produktion"], + "required_event_types": ["technical_finding"], + "outcome": { + "required": true, + "terms": ["kein zusätzlicher", "keine zusätzliche", "unverändert"], + "scope_terms": ["reine produktion", "produktion", "gegenüber virgin"], + "certainties": ["established"] + }, + "actions": {"minimum": 0}, + "unresolved": { + "minimum": 1, + "terms": ["veröffentlich", "freigabe", "energieaudit", "daten"] + } + } + } + ] +} diff --git a/tests/test_evidence_observation_experiment.py b/tests/test_evidence_observation_experiment.py new file mode 100644 index 0000000..a9ec614 --- /dev/null +++ b/tests/test_evidence_observation_experiment.py @@ -0,0 +1,168 @@ +import json +import tempfile +import unittest +from copy import deepcopy +from pathlib import Path +from unittest.mock import patch + +from src.meeting_lab.evidence_observations.experiment import ( + SCHEMA_VERSION, + ObservationValidationError, + build_ollama_payload, + load_fixture, + parse_model_json, + run_case, + validate_observations, +) + + +class EvidenceObservationExperimentTests(unittest.TestCase): + def setUp(self) -> None: + self.case = { + "case_id": "test_case", + "description": "Validator fixture.", + "subject_id": "subject_test", + "subject": "Prüfung der Messdaten", + "evidence": [ + {"evidence_id": "e1", "text": "Nina, prüfst du die Daten?"}, + {"evidence_id": "e2", "text": "Ja, ich prüfe sie."}, + ], + "expected_observations": [], + } + self.output = { + "schema_version": SCHEMA_VERSION, + "subject_id": "subject_test", + "subject": "Prüfung der Messdaten", + "observations": [self.observation()], + } + self.case["expected_observations"] = deepcopy(self.output["observations"]) + + def observation(self, **updates): + value = { + "observation_id": "obs_1", + "evidence_id": "e1", + "content": "Nina wird um Prüfung gebeten.", + "target": "discussion_subject", + "relation": "none", + "modality": "interpersonal_request", + "temporality": "future", + "evaluation": "none", + "agreement": "none", + "responsibility": "named", + "person": "Nina", + "uncertainty": "absent", + "clarification_need": "none", + "scope": "absent", + } + value.update(updates) + return value + + def test_valid_observation_and_discussion_subject_target(self): + self.assertIs(validate_observations(self.output, self.case), self.output) + + def test_multiple_observations_from_one_evidence_unit_and_observation_target(self): + second = self.observation( + observation_id="obs_2", target="obs_1", relation="supports" + ) + self.output["observations"].append(second) + validate_observations(self.output, self.case) + + def test_plural_target_is_allowed_for_joint_reference(self): + self.output["observations"].extend( + [ + self.observation(observation_id="obs_2"), + self.observation( + observation_id="obs_3", + target=["obs_1", "obs_2"], + relation="qualifies", + ), + ] + ) + validate_observations(self.output, self.case) + + def test_unknown_evidence_reference_is_rejected(self): + self.output["observations"][0]["evidence_id"] = "missing" + with self.assertRaisesRegex(ObservationValidationError, "unknown evidence"): + validate_observations(self.output, self.case) + + def test_unknown_observation_target_is_rejected(self): + self.output["observations"][0]["target"] = "obs_9" + with self.assertRaisesRegex(ObservationValidationError, "unknown or later"): + validate_observations(self.output, self.case) + + def test_invalid_relation_is_rejected(self): + self.output["observations"][0]["relation"] = "causes" + with self.assertRaisesRegex(ObservationValidationError, "relation is invalid"): + validate_observations(self.output, self.case) + + def test_invalid_modality_is_rejected(self): + self.output["observations"][0]["modality"] = "proposal" + with self.assertRaisesRegex(ObservationValidationError, "modality is invalid"): + validate_observations(self.output, self.case) + + def test_invalid_responsibility_person_combinations_are_rejected(self): + self.output["observations"][0].update(responsibility="none", person="Nina") + with self.assertRaisesRegex(ObservationValidationError, "person must be JSON null"): + validate_observations(self.output, self.case) + + self.output["observations"][0].update(responsibility="accepted", person=None) + with self.assertRaisesRegex(ObservationValidationError, "person must be a non-empty"): + validate_observations(self.output, self.case) + + def test_scope_uses_absent_or_nonempty_evidence_grounded_text(self): + validate_observations(self.output, self.case) + self.output["observations"][0]["scope"] = "bis Freitag" + validate_observations(self.output, self.case) + self.output["observations"][0]["scope"] = None + with self.assertRaisesRegex(ObservationValidationError, "non-empty string"): + validate_observations(self.output, self.case) + + def test_string_null_is_rejected_in_text_fields(self): + self.output["observations"][0]["scope"] = "null" + with self.assertRaisesRegex(ObservationValidationError, "string 'null'"): + validate_observations(self.output, self.case) + + def test_malformed_model_json_is_rejected(self): + with self.assertRaises(json.JSONDecodeError): + parse_model_json("{not json") + + def test_payload_has_exact_live_controls(self): + payload = build_ollama_payload("qwen3.5:9B", "prompt", 16384, 4096) + self.assertIs(payload["think"], False) + self.assertIs(payload["stream"], False) + self.assertEqual(payload["format"], "json") + self.assertEqual(payload["options"]["temperature"], 0) + + def test_fixture_contains_all_nine_cases(self): + cases = load_fixture(Path("tests/gold/evidence_observations_v1/cases.json")) + self.assertEqual(len(cases), 9) + self.assertEqual(cases[0]["case_id"], "a_idea_only") + self.assertEqual(cases[-1]["case_id"], "i_outcome_and_unresolved") + + def test_case_run_preserves_all_artifacts(self): + raw = json.dumps(self.output, ensure_ascii=False) + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) + with patch( + "src.meeting_lab.evidence_observations.experiment.call_ollama", + return_value=(raw, {"model": "qwen3.5:9B"}), + ): + result = run_case( + self.case, root, "http://unused", "qwen3.5:9B", 1, 16384, 4096 + ) + self.assertEqual(result["verdict"], "PASS") + for filename in ( + "gold_input.json", + "gold_expected_observations.json", + "prompt.txt", + "raw_model_response.txt", + "parsed_observations.json", + "validation_result.json", + "ollama_metadata.json", + "evaluation_result.json", + ): + self.assertTrue((root / "test_case" / filename).is_file(), filename) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_evidence_observation_v2_experiment.py b/tests/test_evidence_observation_v2_experiment.py new file mode 100644 index 0000000..5e593c3 --- /dev/null +++ b/tests/test_evidence_observation_v2_experiment.py @@ -0,0 +1,160 @@ +import json +import tempfile +import unittest +from copy import deepcopy +from pathlib import Path +from unittest.mock import patch + +from src.meeting_lab.evidence_observations_v2.experiment import ( + SCHEMA_VERSION, + ObservationValidationError, + build_ollama_payload, + load_fixture, + parse_model_json, + run_case, + validate_observations, +) + + +class EvidenceObservationV2ExperimentTests(unittest.TestCase): + def setUp(self) -> None: + self.case = { + "case_id": "test_case", "description": "Validator fixture.", + "subject_id": "subject_test", "subject": "Prüfung der Messdaten", + "evidence": [ + {"evidence_id": "e1", "text": "Antonius: Nina, prüfst du die Daten?"}, + {"evidence_id": "e2", "text": "Nina: Ja, ich prüfe sie."}, + ], + "expected_observations": [], + } + self.output = { + "schema_version": SCHEMA_VERSION, + "subject_id": self.case["subject_id"], "subject": self.case["subject"], + "observations": [self.observation()], + } + self.case["expected_observations"] = deepcopy(self.output["observations"]) + + def observation(self, **updates): + value = { + "observation_id": "obs_1", "evidence_id": "e1", + "content": "Antonius bittet Nina um eine Prüfung.", "refers_to": None, + "speaker": "Antonius", "named_person": "Nina", "addressee": "Nina", + "self_reference": False, "collective_we": False, + "impersonal_person_reference": False, + "modality": "interpersonal_request", "temporality": "future", + "evaluation": "none", "affirmation": "absent", "negation": "absent", + "determination_statement": "absent", "uncertainty": "absent", + "clarification_need": "none", "qualifier": "bis Freitag", + "limits_target": None, + } + value.update(updates) + return value + + def test_participant_facts_do_not_include_responsibility(self): + validate_observations(self.output, self.case) + observation = self.output["observations"][0] + self.assertEqual(observation["speaker"], "Antonius") + self.assertEqual(observation["named_person"], "Nina") + self.assertEqual(observation["addressee"], "Nina") + self.assertNotIn("responsibility", observation) + + def test_named_person_and_speaker_do_not_imply_any_extra_field(self): + keys = self.output["observations"][0].keys() + self.assertNotIn("person", keys) + self.assertNotIn("agreement", keys) + + def test_self_reference_collective_we_and_impersonal_reference_are_boolean(self): + self.output["observations"][0].update( + self_reference=True, collective_we=True, impersonal_person_reference=True + ) + validate_observations(self.output, self.case) + self.output["observations"][0]["collective_we"] = "true" + with self.assertRaisesRegex(ObservationValidationError, "must be boolean"): + validate_observations(self.output, self.case) + + def test_explicit_affirmation_negation_and_determination(self): + self.output["observations"][0].update( + affirmation="explicit", negation="explicit", determination_statement="present" + ) + validate_observations(self.output, self.case) + + def test_scalar_reference_to_prior_observation(self): + self.output["observations"].append(self.observation( + observation_id="obs_2", evidence_id="e2", refers_to="obs_1", + speaker="Nina", named_person=None, addressee=None, + )) + validate_observations(self.output, self.case) + + def test_array_and_invalid_reference_are_rejected(self): + self.output["observations"][0]["refers_to"] = ["obs_1"] + with self.assertRaisesRegex(ObservationValidationError, "non-empty string"): + validate_observations(self.output, self.case) + self.output["observations"][0]["refers_to"] = "obs_9" + with self.assertRaisesRegex(ObservationValidationError, "unknown or later"): + validate_observations(self.output, self.case) + + def test_qualifier_is_null_or_nonempty_text(self): + self.output["observations"][0]["qualifier"] = None + validate_observations(self.output, self.case) + self.output["observations"][0]["qualifier"] = "" + with self.assertRaisesRegex(ObservationValidationError, "non-empty string"): + validate_observations(self.output, self.case) + + def test_limits_target_must_reference_prior_observation(self): + self.output["observations"].append(self.observation( + observation_id="obs_2", evidence_id="e2", refers_to="obs_1", + limits_target="obs_1", speaker="Nina", named_person=None, addressee=None, + )) + validate_observations(self.output, self.case) + self.output["observations"][1]["limits_target"] = "obs_7" + with self.assertRaisesRegex(ObservationValidationError, "unknown or later"): + validate_observations(self.output, self.case) + + def test_multiple_atomic_observations_may_share_evidence(self): + self.output["observations"].append(self.observation(observation_id="obs_2")) + validate_observations(self.output, self.case) + + def test_string_null_is_rejected(self): + self.output["observations"][0]["named_person"] = "null" + with self.assertRaisesRegex(ObservationValidationError, "string 'null'"): + validate_observations(self.output, self.case) + + def test_malformed_json_is_rejected(self): + with self.assertRaises(json.JSONDecodeError): + parse_model_json("{not json") + + def test_payload_has_exact_live_controls(self): + payload = build_ollama_payload("qwen3.5:9B", "prompt", 16384, 4096) + self.assertFalse(payload["think"]) + self.assertFalse(payload["stream"]) + self.assertEqual(payload["options"]["temperature"], 0) + + def test_fixture_contains_unchanged_a_i_source_evidence(self): + v1 = load_fixture(Path("tests/gold/evidence_observations_v2/cases.json")) + original = json.loads(Path("tests/gold/evidence_observations_v1/cases.json").read_text())["cases"] + self.assertEqual(len(v1), 9) + self.assertEqual( + [[item["text"] for item in case["evidence"]] for case in v1], + [[item["text"] for item in case["evidence"]] for case in original], + ) + + def test_case_run_preserves_all_artifacts(self): + raw = json.dumps(self.output, ensure_ascii=False) + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) + with patch( + "src.meeting_lab.evidence_observations_v2.experiment.call_ollama", + return_value=(raw, {"model": "qwen3.5:9B"}), + ): + result = run_case(self.case, root, "http://unused", "qwen3.5:9B", 1, 16384, 4096) + self.assertEqual(result["verdict"], "PASS") + for filename in ( + "gold_input.json", "gold_expected_observations.json", "prompt.txt", + "raw_model_response.txt", "parsed_observations.json", + "validation_result.json", "ollama_metadata.json", "evaluation_result.json", + ): + self.assertTrue((root / "test_case" / filename).is_file(), filename) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_evidence_observation_v3_experiment.py b/tests/test_evidence_observation_v3_experiment.py new file mode 100644 index 0000000..92a85a9 --- /dev/null +++ b/tests/test_evidence_observation_v3_experiment.py @@ -0,0 +1,114 @@ +import json +import tempfile +import unittest +from pathlib import Path +from unittest.mock import patch + +from src.meeting_lab.evidence_observations_v3.experiment import ( + SCHEMA_VERSION, + ObservationValidationError, + build_ollama_payload, + load_fixture, + parse_model_json, + run_case, + validate_observations, +) + + +class EvidenceObservationV3ExperimentTests(unittest.TestCase): + def setUp(self) -> None: + self.case = { + "case_id": "test", "description": "Minimal fixture.", + "subject_id": "subject_test", "subject": "Messdatenprüfung", + "evidence": [ + {"evidence_id": "e1", "text": "Antonius: Nina, prüfst du die Messdaten?"}, + {"evidence_id": "e2", "text": "Nina: Ja, ich prüfe sie bis Freitag."}, + ], + "semantic_requirements": ["Request and response survive."], + } + self.output = { + "schema_version": SCHEMA_VERSION, "subject_id": "subject_test", + "subject": "Messdatenprüfung", "observations": [self.observation()], + } + + def observation(self, **updates): + value = { + "observation_id": "obs_1", "evidence_id": "e1", + "content": "Antonius fragt Nina, ob sie die Messdaten prüft.", + "speaker": "Antonius", "named_person": "Nina", "addressee": "Nina", + } + value.update(updates) + return value + + def test_minimal_schema_is_valid(self): + self.assertIs(validate_observations(self.output, self.case), self.output) + self.assertEqual(set(self.output["observations"][0]), { + "observation_id", "evidence_id", "content", "speaker", "named_person", "addressee" + }) + + def test_unknown_semantic_field_is_rejected(self): + self.output["observations"][0]["modality"] = "factual" + with self.assertRaisesRegex(ObservationValidationError, "unknown keys"): + validate_observations(self.output, self.case) + + def test_unknown_evidence_is_rejected(self): + self.output["observations"][0]["evidence_id"] = "e9" + with self.assertRaisesRegex(ObservationValidationError, "unknown evidence"): + validate_observations(self.output, self.case) + + def test_speaker_must_match_evidence(self): + self.output["observations"][0]["speaker"] = "Nina" + with self.assertRaisesRegex(ObservationValidationError, "match evidence speaker"): + validate_observations(self.output, self.case) + + def test_named_person_does_not_add_responsibility(self): + validate_observations(self.output, self.case) + self.assertNotIn("responsibility", self.output["observations"][0]) + + def test_addressee_does_not_add_assignment(self): + validate_observations(self.output, self.case) + self.assertNotIn("action_item", self.output["observations"][0]) + + def test_nonexplicit_person_is_rejected(self): + self.output["observations"][0]["named_person"] = "Martin" + with self.assertRaisesRegex(ObservationValidationError, "not an explicit person"): + validate_observations(self.output, self.case) + + def test_multiple_atomic_observations_can_share_evidence(self): + self.output["observations"].append(self.observation(observation_id="obs_2")) + validate_observations(self.output, self.case) + + def test_malformed_json_is_rejected(self): + with self.assertRaises(json.JSONDecodeError): + parse_model_json("{bad json") + + def test_payload_controls_are_fixed(self): + payload = build_ollama_payload("qwen3.5:9B", "prompt", 16384, 4096) + self.assertFalse(payload["think"]) + self.assertEqual(payload["options"]["temperature"], 0) + + def test_fixture_reuses_exact_v2_evidence(self): + v3 = load_fixture(Path("tests/gold/evidence_observations_v3/cases.json")) + v2 = json.loads(Path("tests/gold/evidence_observations_v2/cases.json").read_text())["cases"] + self.assertEqual([case["evidence"] for case in v3], [case["evidence"] for case in v2]) + + def test_case_run_preserves_persistent_artifact_set(self): + raw = json.dumps(self.output, ensure_ascii=False) + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) + with patch( + "src.meeting_lab.evidence_observations_v3.experiment.call_ollama", + return_value=(raw, {"model": "qwen3.5:9B"}), + ): + result = run_case(self.case, root, "http://unused", "qwen3.5:9B", 1, 16384, 4096) + self.assertTrue(result["structurally_valid"]) + for filename in ( + "source_evidence.json", "gold_semantic_requirements.json", "prompt.txt", + "raw_model_response.txt", "parsed_observations.json", + "structural_validation.json", "ollama_metadata.json", + ): + self.assertTrue((root / "test" / filename).is_file(), filename) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_semantic_synthesis_experiment.py b/tests/test_semantic_synthesis_experiment.py new file mode 100644 index 0000000..a21c255 --- /dev/null +++ b/tests/test_semantic_synthesis_experiment.py @@ -0,0 +1,250 @@ +import json +import tempfile +import unittest +from pathlib import Path +from unittest.mock import patch + +from src.meeting_lab.semantic_synthesis.experiment import ( + SCHEMA_VERSION, + SynthesisValidationError, + build_ollama_payload, + evaluate_synthesis, + load_fixture, + run_case, + validate_bundle, + validate_synthesis, +) + + +class SemanticSynthesisExperimentTests(unittest.TestCase): + def setUp(self) -> None: + self.case = { + "case_id": "case_1", + "description": "Known subject test.", + "subject_id": "subject_1", + "subject": "Prüfung der Messdaten", + "evidence": [ + {"evidence_id": "e1", "text": "Nina übernimmt die Prüfung."}, + {"evidence_id": "e2", "text": "Die Freigabe bleibt offen."}, + ], + "allowed_responsible": ["Nina"], + "expected": { + "event_type_minimums": {"proposal": 1}, + "allowed_event_types": ["proposal"], + "event_evidence_ids": ["e1"], + "outcome": { + "required": True, + "statuses": ["established"], + "terms": ["prüfung"], + "scope_terms": ["messdaten"], + "evidence_ids": ["e1"], + }, + "actions": { + "count": 1, + "terms": ["prüfung"], + "responsible": "Nina", + "due_terms": [], + "evidence_ids": ["e1"], + }, + "unresolved_issues": { + "count": 1, + "terms": ["freigabe"], + "evidence_ids": ["e2"], + }, + }, + } + + def valid_output(self): + return { + "schema_version": SCHEMA_VERSION, + "subject_id": "subject_1", + "subject": "Prüfung der Messdaten", + "events": [ + { + "type": "proposal", + "text": "Die Prüfung wird vorgeschlagen.", + "evidence_ids": ["e1"], + } + ], + "outcome": { + "status": "established", + "text": "Die Prüfung wird übernommen.", + "scope": "Prüfung der Messdaten", + "evidence_ids": ["e1"], + }, + "actions": [ + { + "text": "Prüfung der Messdaten durchführen.", + "responsible": "Nina", + "due": None, + "evidence_ids": ["e1"], + } + ], + "unresolved_issues": [ + { + "text": "Die Freigabe bleibt offen.", + "evidence_ids": ["e2"], + } + ], + } + + def test_bundle_validation_accepts_fixed_subject_and_complete_evidence(self): + self.assertIs(validate_bundle(self.case), self.case) + + def test_bundle_validation_rejects_duplicate_evidence_ids(self): + case = dict(self.case) + case["evidence"] = self.case["evidence"] * 2 + with self.assertRaisesRegex(SynthesisValidationError, "duplicate evidence ID"): + validate_bundle(case) + + def test_sparse_absence_uses_empty_arrays_and_omitted_outcome(self): + output = { + "schema_version": SCHEMA_VERSION, + "subject_id": "subject_1", + "subject": "Prüfung der Messdaten", + "events": [], + "actions": [], + "unresolved_issues": [], + } + + self.assertIs(validate_synthesis(output, self.case), output) + + def test_outcome_null_is_rejected_but_omission_is_allowed(self): + output = self.valid_output() + output["outcome"] = None + with self.assertRaisesRegex(SynthesisValidationError, "omit it when absent"): + validate_synthesis(output, self.case) + + def test_required_arrays_must_exist(self): + for field in ("events", "actions", "unresolved_issues"): + with self.subTest(field=field): + output = self.valid_output() + del output[field] + with self.assertRaisesRegex(SynthesisValidationError, "missing required"): + validate_synthesis(output, self.case) + + def test_fixed_subject_identity_cannot_change(self): + output = self.valid_output() + output["subject"] = "Different subject" + with self.assertRaisesRegex(SynthesisValidationError, "changed fixed subject"): + validate_synthesis(output, self.case) + + def test_unknown_evidence_id_is_rejected_in_every_structure(self): + mutations = ( + lambda output: output["events"][0].update(evidence_ids=["unknown"]), + lambda output: output["outcome"].update(evidence_ids=["unknown"]), + lambda output: output["actions"][0].update(evidence_ids=["unknown"]), + lambda output: output["unresolved_issues"][0].update( + evidence_ids=["unknown"] + ), + ) + for mutate in mutations: + output = self.valid_output() + mutate(output) + with self.assertRaisesRegex(SynthesisValidationError, "unknown evidence ID"): + validate_synthesis(output, self.case) + + def test_duplicate_evidence_reference_is_rejected(self): + output = self.valid_output() + output["events"][0]["evidence_ids"] = ["e1", "e1"] + with self.assertRaisesRegex(SynthesisValidationError, "duplicate evidence ID"): + validate_synthesis(output, self.case) + + def test_responsibility_must_be_allowed_or_json_null(self): + output = self.valid_output() + output["actions"][0]["responsible"] = None + validate_synthesis(output, self.case) + + output["actions"][0]["responsible"] = "Martin" + with self.assertRaisesRegex(SynthesisValidationError, "not allowed"): + validate_synthesis(output, self.case) + + def test_string_null_is_rejected(self): + output = self.valid_output() + output["actions"][0]["due"] = "null" + with self.assertRaisesRegex(SynthesisValidationError, "JSON null"): + validate_synthesis(output, self.case) + + def test_outcome_action_and_unresolved_structures_are_strict(self): + for field, target in ( + ("extra", lambda output: output["outcome"]), + ("extra", lambda output: output["actions"][0]), + ("extra", lambda output: output["unresolved_issues"][0]), + ): + output = self.valid_output() + target(output)[field] = "not allowed" + with self.assertRaisesRegex(SynthesisValidationError, "unknown keys"): + validate_synthesis(output, self.case) + + def test_evaluator_passes_complete_semantics(self): + result = evaluate_synthesis(self.valid_output(), self.case["expected"]) + self.assertEqual(result["verdict"], "PASS") + + def test_evaluator_treats_invented_action_as_critical(self): + output = self.valid_output() + expected = dict(self.case["expected"]) + expected["actions"] = {"count": 0} + result = evaluate_synthesis(output, expected) + self.assertEqual(result["verdict"], "FAIL") + self.assertIn("action_count", result["critical_failures"]) + + def test_ollama_payload_is_bounded_and_has_required_controls(self): + payload = build_ollama_payload("qwen3.5:9B", "prompt", 8192, 2048) + self.assertEqual(payload["format"], "json") + self.assertIs(payload["think"], False) + self.assertIs(payload["stream"], False) + self.assertEqual(payload["options"]["temperature"], 0) + self.assertEqual(payload["options"]["num_ctx"], 8192) + self.assertEqual(payload["options"]["num_predict"], 2048) + + def test_fixture_contains_all_nine_isolation_cases(self): + cases = load_fixture(Path("tests/gold/semantic_synthesis_isolation/cases.json")) + self.assertEqual( + [case["case_id"] for case in cases], + [ + "a_idea_only", + "b_multiple_options", + "c_unaccepted_proposal", + "d_proposal_with_objection", + "e_rejected_alternative", + "f_trial_only_acceptance", + "g_no_decision", + "h_resulting_action", + "i_outcome_and_unresolved", + ], + ) + + def test_case_run_preserves_all_inspection_artifacts(self): + raw = json.dumps(self.valid_output(), ensure_ascii=False) + metadata = {"elapsed_seconds": 0.01} + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) + with patch( + "src.meeting_lab.semantic_synthesis.experiment.call_ollama", + return_value=(raw, metadata), + ): + result = run_case( + self.case, + root, + "http://unused", + "qwen3.5:9B", + 1, + 8192, + 2048, + ) + self.assertEqual(result["verdict"], "PASS") + case_dir = root / "case_1" + for filename in ( + "gold_input.json", + "prompt.txt", + "raw_model_response.txt", + "parsed_response.json", + "ollama_metadata.json", + "validation_result.json", + "evaluation_result.json", + ): + self.assertTrue((case_dir / filename).exists(), filename) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_topic_reconstruction_experiment.py b/tests/test_topic_reconstruction_experiment.py new file mode 100644 index 0000000..e773a81 --- /dev/null +++ b/tests/test_topic_reconstruction_experiment.py @@ -0,0 +1,292 @@ +import json +import tempfile +import unittest +from pathlib import Path +from unittest.mock import patch + +from src.meeting_lab.topic_reconstruction.experiment import ( + ReconstructionValidationError, + SCHEMA_VERSION, + build_ollama_payload, + evaluate_reconstruction, + run_case, + validate_evidence_units, + validate_reconstruction, +) + + +class TopicReconstructionExperimentTests(unittest.TestCase): + def setUp(self) -> None: + self.evidence = [ + {"evidence_id": "e1", "text": "Eine Variante wird vorgeschlagen."}, + {"evidence_id": "e2", "text": "Die Variante wird nur getestet."}, + {"evidence_id": "e3", "text": "Nina übernimmt die Prüfung."}, + {"evidence_id": "e4", "text": "Die Freigabe bleibt ungeklärt."}, + ] + + def valid_output(self): + return { + "schema_version": SCHEMA_VERSION, + "subjects": [ + { + "subject_id": "subject_1", + "title": "Versuch mit der Variante", + "evidence_refs": ["e1", "e2", "e3", "e4"], + "development": [ + { + "event_id": "event_1", + "type": "proposal", + "text": "Die Variante wurde für einen Versuch vorgeschlagen.", + "evidence_refs": ["e1"], + } + ], + "outcome": { + "text": "Die Variante wird getestet.", + "scope": "Nur für den Versuch, nicht als endgültige Lösung.", + "certainty": "established", + "evidence_refs": ["e2"], + }, + "actions": [ + { + "action_id": "action_1", + "text": "Die Variante prüfen.", + "responsible": "Nina", + "deadline": None, + "evidence_refs": ["e3"], + } + ], + "unresolved_issues": [ + { + "issue_id": "issue_1", + "text": "Die Freigabe ist ungeklärt.", + "evidence_refs": ["e4"], + } + ], + } + ], + } + + def test_schema_validation_accepts_sparse_subject(self): + output = { + "schema_version": SCHEMA_VERSION, + "subjects": [ + { + "subject_id": "subject_1", + "title": "Geometrie", + "evidence_refs": ["e1"], + } + ], + } + + self.assertIs(validate_reconstruction(output, self.evidence), output) + + def test_schema_validation_accepts_complete_structures(self): + output = self.valid_output() + + self.assertIs(validate_reconstruction(output, self.evidence), output) + + def test_every_semantic_structure_requires_evidence_traceability(self): + structures = [ + ("subject", lambda data: data["subjects"][0].update(evidence_refs=[])), + ( + "event", + lambda data: data["subjects"][0]["development"][0].update( + evidence_refs=[] + ), + ), + ( + "outcome", + lambda data: data["subjects"][0]["outcome"].update(evidence_refs=[]), + ), + ( + "action", + lambda data: data["subjects"][0]["actions"][0].update( + evidence_refs=[] + ), + ), + ( + "unresolved", + lambda data: data["subjects"][0]["unresolved_issues"][0].update( + evidence_refs=[] + ), + ), + ] + for name, mutate in structures: + with self.subTest(name=name): + data = self.valid_output() + mutate(data) + with self.assertRaisesRegex( + ReconstructionValidationError, "non-empty list" + ): + validate_reconstruction(data, self.evidence) + + def test_unknown_evidence_reference_is_rejected(self): + output = self.valid_output() + output["subjects"][0]["outcome"]["evidence_refs"] = ["e999"] + + with self.assertRaisesRegex( + ReconstructionValidationError, "unknown evidence ID: e999" + ): + validate_reconstruction(output, self.evidence) + + def test_duplicate_semantic_identifier_is_rejected(self): + output = self.valid_output() + output["subjects"][0]["actions"][0]["action_id"] = "event_1" + + with self.assertRaisesRegex( + ReconstructionValidationError, "duplicate identifier: event_1" + ): + validate_reconstruction(output, self.evidence) + + def test_duplicate_input_evidence_identifier_is_rejected(self): + evidence = self.evidence + [ + {"evidence_id": "e1", "text": "Duplicate source."} + ] + + with self.assertRaisesRegex( + ReconstructionValidationError, "duplicate input evidence identifier" + ): + validate_evidence_units(evidence) + + def test_empty_subjects_are_rejected(self): + output = {"schema_version": SCHEMA_VERSION, "subjects": []} + + with self.assertRaisesRegex( + ReconstructionValidationError, "subjects must be a non-empty list" + ): + validate_reconstruction(output, self.evidence) + + def test_blank_subject_title_is_rejected(self): + output = self.valid_output() + output["subjects"][0]["title"] = " " + + with self.assertRaisesRegex( + ReconstructionValidationError, "title must be a non-empty string" + ): + validate_reconstruction(output, self.evidence) + + def test_empty_optional_structures_must_be_omitted(self): + for field, value in ( + ("development", []), + ("outcome", None), + ("actions", []), + ("unresolved_issues", []), + ): + with self.subTest(field=field): + output = { + "schema_version": SCHEMA_VERSION, + "subjects": [ + { + "subject_id": "subject_1", + "title": "Subject", + "evidence_refs": ["e1"], + field: value, + } + ], + } + with self.assertRaises(ReconstructionValidationError): + validate_reconstruction(output, self.evidence) + + def test_outcome_requires_scope_and_valid_certainty(self): + output = self.valid_output() + output["subjects"][0]["outcome"]["scope"] = "" + with self.assertRaisesRegex(ReconstructionValidationError, "scope"): + validate_reconstruction(output, self.evidence) + + output = self.valid_output() + output["subjects"][0]["outcome"]["certainty"] = "accepted_forever" + with self.assertRaisesRegex(ReconstructionValidationError, "certainty"): + validate_reconstruction(output, self.evidence) + + def test_action_nullable_fields_and_unresolved_structure_are_strict(self): + output = self.valid_output() + output["subjects"][0]["actions"][0]["responsible"] = None + validate_reconstruction(output, self.evidence) + + output["subjects"][0]["unresolved_issues"][0]["extra"] = "invented" + with self.assertRaisesRegex(ReconstructionValidationError, "unknown keys"): + validate_reconstruction(output, self.evidence) + + def test_action_nullable_fields_reject_string_null(self): + output = self.valid_output() + output["subjects"][0]["actions"][0]["responsible"] = "null" + + with self.assertRaisesRegex(ReconstructionValidationError, "JSON null"): + validate_reconstruction(output, self.evidence) + + def test_ollama_payload_is_bounded_and_disables_thinking(self): + payload = build_ollama_payload("qwen3.5:9B", "prompt", 16384, 4096) + + self.assertEqual(payload["model"], "qwen3.5:9B") + self.assertEqual(payload["format"], "json") + self.assertIs(payload["stream"], False) + self.assertIs(payload["think"], False) + self.assertEqual(payload["options"]["temperature"], 0) + self.assertEqual(payload["options"]["num_ctx"], 16384) + self.assertEqual(payload["options"]["num_predict"], 4096) + + def test_evaluator_marks_invented_action_as_critical_failure(self): + output = self.valid_output() + expected = { + "subject_count": 1, + "subject_terms": ["variante"], + "required_event_types": ["proposal"], + "outcome": { + "required": True, + "terms": ["getestet"], + "scope_terms": ["nur"], + "certainties": ["established"], + }, + "actions": {"minimum": 0}, + "unresolved": {"minimum": 1, "terms": ["freigabe"]}, + } + + result = evaluate_reconstruction(output, expected) + + self.assertEqual(result["verdict"], "FAIL") + self.assertIn("action_count", result["critical_failures"]) + + def test_validation_failure_preserves_inspection_artifacts(self): + invalid = self.valid_output() + invalid["subjects"][0]["outcome"]["evidence_refs"] = ["unknown"] + raw = json.dumps(invalid, ensure_ascii=False) + case = { + "case_id": "artifact_case", + "description": "Artifact preservation test.", + "evidence_units": self.evidence, + "expected": {}, + } + metadata = {"elapsed_seconds": 0.01} + + with tempfile.TemporaryDirectory() as temporary: + root = Path(temporary) + with patch( + "src.meeting_lab.topic_reconstruction.experiment.call_ollama", + return_value=(raw, metadata), + ): + result = run_case( + case, + root, + "http://unused", + "qwen3.5:9B", + 1, + 1024, + 256, + ) + + case_dir = root / "artifact_case" + self.assertEqual(result["verdict"], "FAIL") + self.assertIn("schema_validation", result["critical_failures"]) + self.assertTrue((case_dir / "input.json").exists()) + self.assertTrue((case_dir / "prompt.txt").exists()) + self.assertTrue((case_dir / "raw_model_response.txt").exists()) + self.assertTrue((case_dir / "parsed_output.json").exists()) + self.assertTrue((case_dir / "ollama_metadata.json").exists()) + failure = json.loads( + (case_dir / "validation_failure.json").read_text(encoding="utf-8") + ) + self.assertEqual(failure["error_type"], "ReconstructionValidationError") + + +if __name__ == "__main__": + unittest.main()