Add evidence-near semantic architecture experiments

Record the V1-V3 experiments and accept the minimal semantic-preservation first stage.
This commit is contained in:
2026-08-19 15:46:22 +02:00
parent bcb197a908
commit 18beb3385f
29 changed files with 4542 additions and 0 deletions
@@ -0,0 +1,147 @@
{
"cases": [
{
"case_id": "a_idea_only",
"description": "Possible geometry optimization without commitment.",
"subject_id": "subject_a",
"subject": "Optimierung der Geometrie",
"evidence": [{"evidence_id": "e1", "text": "Martin: Die Geometrie kann man vielleicht noch optimieren. Dann würde man mal gucken, was herauskommt."}],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Die Geometrie kann vielleicht optimiert werden.","target":"discussion_subject","relation":"none","modality":"possible","temporality":"future","evaluation":"positive","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"none","scope":"absent"},
{"observation_id":"obs_2","evidence_id":"e1","content":"Danach könnte betrachtet werden, was herauskommt.","target":"obs_1","relation":"qualifies","modality":"suggested","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"implicit","scope":"nach der Optimierung|danach"}
]
},
{
"case_id": "b_multiple_options",
"description": "Two alternatives for insufficient grid strength.",
"subject_id": "subject_b",
"subject": "Umgang mit unzureichender Festigkeit des 40-40-Gitters",
"evidence": [
{"evidence_id":"e1","text":"Martin: Die Festigkeit reicht für das 40-40-Gitter noch nicht aus."},
{"evidence_id":"e2","text":"Martin: Man könnte mehr Masse für die gleiche Festigkeit einsetzen."},
{"evidence_id":"e3","text":"Martin: Oder wir verkaufen es nicht als 40-40-Gitter, sondern machen ein 20-20 daraus. Das wären die zwei Ansätze."}
],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Die Festigkeit reicht noch nicht aus.","target":"discussion_subject","relation":"none","modality":"factual","temporality":"existing","evaluation":"negative","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"40-40-Gitter|40-40"},
{"observation_id":"obs_2","evidence_id":"e2","content":"Mehr Masse könnte für die gleiche Festigkeit eingesetzt werden.","target":"obs_1","relation":"qualifies","modality":"possible","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"mehr Masse|gleiche Festigkeit"},
{"observation_id":"obs_3","evidence_id":"e3","content":"Das Produkt könnte als 20-20 statt 40-40 ausgeführt werden.","target":"obs_1","relation":"qualifies","modality":"suggested","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"20-20|statt 40-40"},
{"observation_id":"obs_4","evidence_id":"e3","content":"Die vorherigen Möglichkeiten sind die zwei Ansätze.","target":["obs_2","obs_3"],"relation":"qualifies","modality":"factual","temporality":"existing","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"absent"}
]
},
{
"case_id": "c_unaccepted_proposal",
"description": "Suggested Textor contact without established work.",
"subject_id": "subject_c",
"subject": "Erneute Kontaktaufnahme mit Dirk Textor zur Einschätzung",
"evidence": [
{"evidence_id":"e1","text":"Tim: Ich würde vielleicht Dirk Textor noch einmal kontaktieren und fragen, wie er das einschätzt."},
{"evidence_id":"e2","text":"Tim: Das kann man ja mit ihm einfach noch einmal rückkoppeln."}
],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Tim erwägt, Dirk Textor erneut zu kontaktieren und nach seiner Einschätzung zu fragen.","target":"discussion_subject","relation":"none","modality":"suggested","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"none","scope":"Dirk Textors Einschätzung|erneut kontaktieren"},
{"observation_id":"obs_2","evidence_id":"e2","content":"Eine erneute Rückkopplung mit Dirk Textor ist möglich.","target":"obs_1","relation":"supports","modality":"possible","temporality":"future","evaluation":"positive","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"Rückkopplung mit Dirk Textor|mit ihm"}
]
},
{
"case_id": "d_proposal_with_objection",
"description": "Washing possibility and explicit energy disadvantage.",
"subject_id": "subject_d",
"subject": "Waschen des Materials vor der weiteren Verarbeitung",
"evidence": [
{"evidence_id":"e1","text":"Antonius: Man könnte das Material vor der weiteren Verarbeitung waschen."},
{"evidence_id":"e2","text":"Martin: Ob sich das lohnt, weiß ich nicht. Waschen heißt nass machen und wieder trocknen; das ist ein wahnsinniger Energieaufwand."}
],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Das Material könnte gewaschen werden.","target":"discussion_subject","relation":"none","modality":"possible","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"vor der weiteren Verarbeitung"},
{"observation_id":"obs_2","evidence_id":"e2","content":"Martin weiß nicht, ob sich das Waschen lohnt.","target":"obs_1","relation":"qualifies","modality":"factual","temporality":"existing","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"none","scope":"Nutzen des Waschens|ob es sich lohnt"},
{"observation_id":"obs_3","evidence_id":"e2","content":"Waschen umfasst Nassmachen und erneutes Trocknen.","target":"obs_1","relation":"qualifies","modality":"factual","temporality":"existing","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"Waschprozess|Nassmachen und Trocknen"},
{"observation_id":"obs_4","evidence_id":"e2","content":"Waschen und Trocknen verursachen einen sehr hohen Energieaufwand.","target":"obs_1","relation":"opposes","modality":"factual","temporality":"existing","evaluation":"negative","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"Energieaufwand des Waschens|Waschen und Trocknen"}
]
},
{
"case_id": "e_rejected_alternative",
"description": "Explicit rejection followed by confirmation of that rejection.",
"subject_id": "subject_e",
"subject": "Zusammenarbeit mit Dr. Schlummer für Versuche",
"evidence": [
{"evidence_id":"e1","text":"Antonius: Das Angebot von Dr. Schlummer für die Versuche kostet 30.000 Euro."},
{"evidence_id":"e2","text":"Tim: Dann haben wir gesagt: Nein, die Zusammenarbeit mit Dr. Schlummer machen wir nicht."},
{"evidence_id":"e3","text":"Antonius: Ja, das ist entschieden."}
],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Das Angebot kostet 30.000 Euro.","target":"discussion_subject","relation":"none","modality":"factual","temporality":"existing","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"Angebot für die Versuche|30.000 Euro"},
{"observation_id":"obs_2","evidence_id":"e2","content":"Die Zusammenarbeit mit Dr. Schlummer wird nicht durchgeführt.","target":"discussion_subject","relation":"none","modality":"committed","temporality":"future","evaluation":"none","agreement":"rejected","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"Zusammenarbeit für die Versuche|Dr. Schlummer"},
{"observation_id":"obs_3","evidence_id":"e3","content":"Die vorherige Ablehnung ist entschieden.","target":"obs_2","relation":"supports","modality":"factual","temporality":"completed","evaluation":"none","agreement":"accepted","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"absent"}
]
},
{
"case_id": "f_trial_only_acceptance",
"description": "Acceptance limited to a 20-metre trial.",
"subject_id": "subject_f",
"subject": "20-Prozent-Variante im Versuch am kleinen Extruder",
"evidence": [
{"evidence_id":"e1","text":"Martin: Wir könnten die 20-Prozent-Variante am kleinen Extruder nachstellen."},
{"evidence_id":"e2","text":"Tim: Ja, wir testen 20 Meter dieser Variante beim nächsten Versuch."},
{"evidence_id":"e3","text":"Tim: Das ist nur ein Versuch; damit ist die Variante noch nicht als Serienlösung festgelegt."}
],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Die 20-Prozent-Variante könnte am kleinen Extruder nachgestellt werden.","target":"discussion_subject","relation":"none","modality":"possible","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"kleiner Extruder"},
{"observation_id":"obs_2","evidence_id":"e2","content":"20 Meter der Variante werden beim nächsten Versuch getestet.","target":"obs_1","relation":"supports","modality":"committed","temporality":"future","evaluation":"none","agreement":"accepted","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"20 Meter beim nächsten Versuch|20 Meter"},
{"observation_id":"obs_3","evidence_id":"e3","content":"Die Zusage gilt nur für einen Versuch.","target":"obs_2","relation":"limits_scope","modality":"factual","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"nur ein Versuch|Versuch"},
{"observation_id":"obs_4","evidence_id":"e3","content":"Die Variante ist noch nicht als Serienlösung festgelegt.","target":"discussion_subject","relation":"qualifies","modality":"factual","temporality":"existing","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"implicit","scope":"Serienlösung|finale Produktion"}
]
},
{
"case_id": "g_no_decision",
"description": "Preference, alternative, and impersonal checking need without decision.",
"subject_id": "subject_g",
"subject": "Reale Recyclinganlage oder Technikum und verfügbarer Reinigungsansatz",
"evidence": [
{"evidence_id":"e1","text":"Martin: Eine reale Recyclinganlage hätte das Risiko, dass wir kontaminiertes Material zurückbekommen."},
{"evidence_id":"e2","text":"Martin: Ich würde nicht in eine reale Anlage gehen. Wenn überhaupt, können wir über ein Technikum reden."},
{"evidence_id":"e3","text":"Tim: Man müsste zunächst prüfen, welcher Reinigungsansatz überhaupt verfügbar ist."}
],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Eine reale Recyclinganlage birgt das Risiko kontaminierten Rückmaterials.","target":"discussion_subject","relation":"none","modality":"possible","temporality":"future","evaluation":"negative","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"none","scope":"reale Recyclinganlage|kontaminiertes Material"},
{"observation_id":"obs_2","evidence_id":"e2","content":"Martin würde nicht in eine reale Anlage gehen.","target":"discussion_subject","relation":"opposes","modality":"suggested","temporality":"future","evaluation":"negative","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"Martins persönliche Präferenz|reale Anlage"},
{"observation_id":"obs_3","evidence_id":"e2","content":"Ein Technikum bleibt als bedingte Möglichkeit im Gespräch.","target":"discussion_subject","relation":"none","modality":"possible","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"none","scope":"wenn überhaupt|Technikum"},
{"observation_id":"obs_4","evidence_id":"e3","content":"Zunächst muss geprüft werden, welcher Reinigungsansatz verfügbar ist.","target":"discussion_subject","relation":"qualifies","modality":"impersonal_necessity","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"explicit","scope":"zunächst|verfügbarer Reinigungsansatz"}
]
},
{
"case_id": "h_resulting_action",
"description": "Interpersonal request followed by accepted responsibility.",
"subject_id": "subject_h",
"subject": "Prüfung der Messdaten bis Freitag",
"evidence": [
{"evidence_id":"e1","text":"Antonius: Nina, übernimmst du die Prüfung der Messdaten bis Freitag?"},
{"evidence_id":"e2","text":"Nina: Ja, ich übernehme die Prüfung bis Freitag."}
],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Antonius bittet Nina um die Prüfung der Messdaten.","target":"discussion_subject","relation":"none","modality":"interpersonal_request","temporality":"future","evaluation":"none","agreement":"none","responsibility":"named","person":"Nina","uncertainty":"absent","clarification_need":"none","scope":"bis Freitag|Freitag"},
{"observation_id":"obs_2","evidence_id":"e2","content":"Nina übernimmt die Prüfung.","target":"obs_1","relation":"supports","modality":"committed","temporality":"future","evaluation":"none","agreement":"accepted","responsibility":"accepted","person":"Nina","uncertainty":"absent","clarification_need":"none","scope":"bis Freitag|Freitag"}
]
},
{
"case_id": "i_outcome_and_unresolved",
"description": "Bounded production finding and unresolved publication information.",
"subject_id": "subject_i",
"subject": "Produktionsaufwand und Veröffentlichung von Energieaudit-Daten",
"evidence": [
{"evidence_id":"e1","text":"Martin: An unserer Anlage gab es bei der reinen Produktion gegenüber dem Standardprodukt praktisch keine Änderung; wir waren nur fünf Grad kälter."},
{"evidence_id":"e2","text":"Antonius: Dann können wir mindestens festhalten: Gegenüber Virgin Material ist bei der reinen Produktion kein zusätzlicher Aufwand notwendig. Davor entsteht natürlich Aufwand."},
{"evidence_id":"e3","text":"Antonius: Welche Daten aus dem Energieaudit dürfen wir veröffentlichen?"},
{"evidence_id":"e4","text":"Martin: Das ist weiterhin ungeklärt. Wir müssen die Freigabe noch klären."}
],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Bei der reinen Produktion gab es praktisch keine Änderung gegenüber dem Standardprodukt.","target":"discussion_subject","relation":"none","modality":"factual","temporality":"completed","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"eigene Anlage, reine Produktion, Standardprodukt|reine Produktion"},
{"observation_id":"obs_2","evidence_id":"e1","content":"Die Produktion erfolgte fünf Grad kälter.","target":"obs_1","relation":"qualifies","modality":"factual","temporality":"completed","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"fünf Grad kälter|5 Grad"},
{"observation_id":"obs_3","evidence_id":"e2","content":"Gegenüber Virgin Material ist bei reiner Produktion kein zusätzlicher Aufwand notwendig.","target":"obs_1","relation":"supports","modality":"factual","temporality":"existing","evaluation":"none","agreement":"accepted","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"reine Produktion gegenüber Virgin Material|Virgin Material"},
{"observation_id":"obs_4","evidence_id":"e2","content":"Vor der reinen Produktion entsteht Aufwand.","target":"obs_3","relation":"limits_scope","modality":"factual","temporality":"existing","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"absent","clarification_need":"none","scope":"vor der reinen Produktion|davor"},
{"observation_id":"obs_5","evidence_id":"e3","content":"Es wird gefragt, welche Energieaudit-Daten veröffentlicht werden dürfen.","target":"discussion_subject","relation":"none","modality":"information_question","temporality":"unspecified","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"explicit","scope":"Veröffentlichung von Energieaudit-Daten|Energieaudit"},
{"observation_id":"obs_6","evidence_id":"e4","content":"Die Veröffentlichungserlaubnis ist weiterhin ungeklärt.","target":"obs_5","relation":"supports","modality":"factual","temporality":"existing","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"explicit","scope":"Veröffentlichungserlaubnis|Freigabe"},
{"observation_id":"obs_7","evidence_id":"e4","content":"Die Freigabe muss noch geklärt werden.","target":"obs_5","relation":"supports","modality":"impersonal_necessity","temporality":"future","evaluation":"none","agreement":"none","responsibility":"none","person":null,"uncertainty":"present","clarification_need":"explicit","scope":"Freigabe zur Veröffentlichung|Freigabe"}
]
}
]
}
@@ -0,0 +1,99 @@
{
"cases": [
{
"case_id": "a_idea_only", "description": "Possible geometry optimization without commitment.",
"subject_id": "subject_a", "subject": "Optimierung der Geometrie",
"evidence": [{"evidence_id": "e1", "text": "Martin: Die Geometrie kann man vielleicht noch optimieren. Dann würde man mal gucken, was herauskommt."}],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Die Geometrie kann vielleicht optimiert werden.","refers_to":null,"speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":true,"modality":"possible","temporality":"future","evaluation":"positive","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"none","qualifier":null,"limits_target":null},
{"observation_id":"obs_2","evidence_id":"e1","content":"Danach könnte betrachtet werden, was herauskommt.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":true,"modality":"suggested","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"implicit","qualifier":"danach","limits_target":null}
]
},
{
"case_id": "b_multiple_options", "description": "Two alternatives for insufficient grid strength.",
"subject_id": "subject_b", "subject": "Umgang mit unzureichender Festigkeit des 40-40-Gitters",
"evidence": [{"evidence_id":"e1","text":"Martin: Die Festigkeit reicht für das 40-40-Gitter noch nicht aus."},{"evidence_id":"e2","text":"Martin: Man könnte mehr Masse für die gleiche Festigkeit einsetzen."},{"evidence_id":"e3","text":"Martin: Oder wir verkaufen es nicht als 40-40-Gitter, sondern machen ein 20-20 daraus. Das wären die zwei Ansätze."}],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Die Festigkeit des 40-40-Gitters reicht noch nicht aus.","refers_to":null,"speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"negative","affirmation":"absent","negation":"explicit","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"40-40-Gitter","limits_target":null},
{"observation_id":"obs_2","evidence_id":"e2","content":"Mehr Masse könnte für die gleiche Festigkeit eingesetzt werden.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":true,"modality":"possible","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"mehr Masse für die gleiche Festigkeit","limits_target":null},
{"observation_id":"obs_3","evidence_id":"e3","content":"Das Produkt könnte als 20-20 statt 40-40 ausgeführt werden.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"suggested","temporality":"future","evaluation":"none","affirmation":"absent","negation":"explicit","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"20-20 statt 40-40","limits_target":null},
{"observation_id":"obs_4","evidence_id":"e3","content":"Die vorherigen Möglichkeiten sind die zwei Ansätze.","refers_to":null,"speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"zwei Ansätze","limits_target":null}
]
},
{
"case_id": "c_unaccepted_proposal", "description": "Suggested Textor contact without established work.",
"subject_id": "subject_c", "subject": "Erneute Kontaktaufnahme mit Dirk Textor zur Einschätzung",
"evidence": [{"evidence_id":"e1","text":"Tim: Ich würde vielleicht Dirk Textor noch einmal kontaktieren und fragen, wie er das einschätzt."},{"evidence_id":"e2","text":"Tim: Das kann man ja mit ihm einfach noch einmal rückkoppeln."}],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Tim erwägt, Dirk Textor erneut zu kontaktieren und nach seiner Einschätzung zu fragen.","refers_to":null,"speaker":"Tim","named_person":"Dirk Textor","addressee":null,"self_reference":true,"collective_we":false,"impersonal_person_reference":false,"modality":"suggested","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"none","qualifier":"erneut; Dirk Textors Einschätzung","limits_target":null},
{"observation_id":"obs_2","evidence_id":"e2","content":"Eine erneute Rückkopplung mit Dirk Textor ist möglich.","refers_to":"obs_1","speaker":"Tim","named_person":"Dirk Textor","addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":true,"modality":"possible","temporality":"future","evaluation":"positive","affirmation":"explicit","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"noch einmal mit ihm","limits_target":null}
]
},
{
"case_id": "d_proposal_with_objection", "description": "Washing possibility and explicit energy disadvantage.",
"subject_id": "subject_d", "subject": "Waschen des Materials vor der weiteren Verarbeitung",
"evidence": [{"evidence_id":"e1","text":"Antonius: Man könnte das Material vor der weiteren Verarbeitung waschen."},{"evidence_id":"e2","text":"Martin: Ob sich das lohnt, weiß ich nicht. Waschen heißt nass machen und wieder trocknen; das ist ein wahnsinniger Energieaufwand."}],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Das Material könnte gewaschen werden.","refers_to":null,"speaker":"Antonius","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":true,"modality":"possible","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"vor der weiteren Verarbeitung","limits_target":null},
{"observation_id":"obs_2","evidence_id":"e2","content":"Martin weiß nicht, ob sich das Waschen lohnt.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":true,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"explicit","determination_statement":"absent","uncertainty":"present","clarification_need":"none","qualifier":"ob es sich lohnt","limits_target":null},
{"observation_id":"obs_3","evidence_id":"e2","content":"Waschen umfasst Nassmachen und erneutes Trocknen.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"nass machen und wieder trocknen","limits_target":null},
{"observation_id":"obs_4","evidence_id":"e2","content":"Waschen und Trocknen verursachen einen sehr hohen Energieaufwand.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"negative","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"Waschen und Trocknen","limits_target":null}
]
},
{
"case_id": "e_rejected_alternative", "description": "Explicit negation followed by confirmation of that determination.",
"subject_id": "subject_e", "subject": "Zusammenarbeit mit Dr. Schlummer für Versuche",
"evidence": [{"evidence_id":"e1","text":"Antonius: Das Angebot von Dr. Schlummer für die Versuche kostet 30.000 Euro."},{"evidence_id":"e2","text":"Tim: Dann haben wir gesagt: Nein, die Zusammenarbeit mit Dr. Schlummer machen wir nicht."},{"evidence_id":"e3","text":"Antonius: Ja, das ist entschieden."}],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Das Angebot von Dr. Schlummer kostet 30.000 Euro.","refers_to":null,"speaker":"Antonius","named_person":"Dr. Schlummer","addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"für die Versuche; 30.000 Euro","limits_target":null},
{"observation_id":"obs_2","evidence_id":"e2","content":"Die Zusammenarbeit mit Dr. Schlummer wird nicht durchgeführt.","refers_to":null,"speaker":"Tim","named_person":"Dr. Schlummer","addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"committed","temporality":"future","evaluation":"none","affirmation":"absent","negation":"explicit","determination_statement":"present","uncertainty":"absent","clarification_need":"none","qualifier":"Zusammenarbeit für die Versuche","limits_target":null},
{"observation_id":"obs_3","evidence_id":"e3","content":"Die vorherige Festlegung ist entschieden.","refers_to":"obs_2","speaker":"Antonius","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"completed","evaluation":"none","affirmation":"explicit","negation":"absent","determination_statement":"present","uncertainty":"absent","clarification_need":"none","qualifier":null,"limits_target":null}
]
},
{
"case_id": "f_trial_only_acceptance", "description": "Affirmed commitment limited to a 20-metre trial.",
"subject_id": "subject_f", "subject": "20-Prozent-Variante im Versuch am kleinen Extruder",
"evidence": [{"evidence_id":"e1","text":"Martin: Wir könnten die 20-Prozent-Variante am kleinen Extruder nachstellen."},{"evidence_id":"e2","text":"Tim: Ja, wir testen 20 Meter dieser Variante beim nächsten Versuch."},{"evidence_id":"e3","text":"Tim: Das ist nur ein Versuch; damit ist die Variante noch nicht als Serienlösung festgelegt."}],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Die 20-Prozent-Variante könnte am kleinen Extruder nachgestellt werden.","refers_to":null,"speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"possible","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"am kleinen Extruder","limits_target":null},
{"observation_id":"obs_2","evidence_id":"e2","content":"20 Meter der Variante werden beim nächsten Versuch getestet.","refers_to":"obs_1","speaker":"Tim","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"committed","temporality":"future","evaluation":"none","affirmation":"explicit","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"20 Meter beim nächsten Versuch","limits_target":null},
{"observation_id":"obs_3","evidence_id":"e3","content":"Die Zusage gilt nur für einen Versuch.","refers_to":"obs_2","speaker":"Tim","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"nur ein Versuch","limits_target":"obs_2"},
{"observation_id":"obs_4","evidence_id":"e3","content":"Die Variante ist noch nicht als Serienlösung festgelegt.","refers_to":"obs_2","speaker":"Tim","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"explicit","determination_statement":"present","uncertainty":"present","clarification_need":"implicit","qualifier":"als Serienlösung","limits_target":null}
]
},
{
"case_id": "g_no_decision", "description": "Preference, alternative, and impersonal checking need without decision.",
"subject_id": "subject_g", "subject": "Reale Recyclinganlage oder Technikum und verfügbarer Reinigungsansatz",
"evidence": [{"evidence_id":"e1","text":"Martin: Eine reale Recyclinganlage hätte das Risiko, dass wir kontaminiertes Material zurückbekommen."},{"evidence_id":"e2","text":"Martin: Ich würde nicht in eine reale Anlage gehen. Wenn überhaupt, können wir über ein Technikum reden."},{"evidence_id":"e3","text":"Tim: Man müsste zunächst prüfen, welcher Reinigungsansatz überhaupt verfügbar ist."}],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Eine reale Recyclinganlage birgt das Risiko kontaminierten Rückmaterials.","refers_to":null,"speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"possible","temporality":"future","evaluation":"negative","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"none","qualifier":"reale Recyclinganlage; kontaminiertes Material","limits_target":null},
{"observation_id":"obs_2","evidence_id":"e2","content":"Martin würde persönlich nicht in eine reale Anlage gehen.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":true,"collective_we":false,"impersonal_person_reference":false,"modality":"suggested","temporality":"future","evaluation":"negative","affirmation":"absent","negation":"explicit","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"reale Anlage","limits_target":null},
{"observation_id":"obs_3","evidence_id":"e2","content":"Ein Technikum bleibt als bedingte Möglichkeit im Gespräch.","refers_to":null,"speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"possible","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"none","qualifier":"wenn überhaupt; Technikum","limits_target":null},
{"observation_id":"obs_4","evidence_id":"e3","content":"Zunächst muss geprüft werden, welcher Reinigungsansatz verfügbar ist.","refers_to":null,"speaker":"Tim","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":true,"modality":"impersonal_necessity","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"explicit","qualifier":"zunächst; verfügbarer Reinigungsansatz","limits_target":null}
]
},
{
"case_id": "h_resulting_action", "description": "Interpersonal request followed by explicit personal acceptance.",
"subject_id": "subject_h", "subject": "Prüfung der Messdaten bis Freitag",
"evidence": [{"evidence_id":"e1","text":"Antonius: Nina, übernimmst du die Prüfung der Messdaten bis Freitag?"},{"evidence_id":"e2","text":"Nina: Ja, ich übernehme die Prüfung bis Freitag."}],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Antonius richtet an Nina die Bitte, die Messdaten zu prüfen.","refers_to":null,"speaker":"Antonius","named_person":"Nina","addressee":"Nina","self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"interpersonal_request","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"bis Freitag","limits_target":null},
{"observation_id":"obs_2","evidence_id":"e2","content":"Nina sagt zu, die Prüfung zu übernehmen.","refers_to":"obs_1","speaker":"Nina","named_person":null,"addressee":null,"self_reference":true,"collective_we":false,"impersonal_person_reference":false,"modality":"committed","temporality":"future","evaluation":"none","affirmation":"explicit","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"bis Freitag","limits_target":null}
]
},
{
"case_id": "i_outcome_and_unresolved", "description": "Bounded production finding and unresolved publication information.",
"subject_id": "subject_i", "subject": "Produktionsaufwand und Veröffentlichung von Energieaudit-Daten",
"evidence": [{"evidence_id":"e1","text":"Martin: An unserer Anlage gab es bei der reinen Produktion gegenüber dem Standardprodukt praktisch keine Änderung; wir waren nur fünf Grad kälter."},{"evidence_id":"e2","text":"Antonius: Dann können wir mindestens festhalten: Gegenüber Virgin Material ist bei der reinen Produktion kein zusätzlicher Aufwand notwendig. Davor entsteht natürlich Aufwand."},{"evidence_id":"e3","text":"Antonius: Welche Daten aus dem Energieaudit dürfen wir veröffentlichen?"},{"evidence_id":"e4","text":"Martin: Das ist weiterhin ungeklärt. Wir müssen die Freigabe noch klären."}],
"expected_observations": [
{"observation_id":"obs_1","evidence_id":"e1","content":"Bei der reinen Produktion gab es praktisch keine Änderung gegenüber dem Standardprodukt.","refers_to":null,"speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"completed","evaluation":"none","affirmation":"absent","negation":"explicit","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"an unserer Anlage; reine Produktion; gegenüber dem Standardprodukt","limits_target":null},
{"observation_id":"obs_2","evidence_id":"e1","content":"Die Produktion erfolgte fünf Grad kälter.","refers_to":"obs_1","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"factual","temporality":"completed","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"fünf Grad kälter","limits_target":null},
{"observation_id":"obs_3","evidence_id":"e2","content":"Gegenüber Virgin Material ist bei reiner Produktion kein zusätzlicher Aufwand notwendig.","refers_to":"obs_1","speaker":"Antonius","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"explicit","determination_statement":"present","uncertainty":"absent","clarification_need":"none","qualifier":"bei reiner Produktion; gegenüber Virgin Material","limits_target":null},
{"observation_id":"obs_4","evidence_id":"e2","content":"Vor der reinen Produktion entsteht Aufwand.","refers_to":"obs_3","speaker":"Antonius","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"absent","clarification_need":"none","qualifier":"davor","limits_target":"obs_3"},
{"observation_id":"obs_5","evidence_id":"e3","content":"Es wird gefragt, welche Energieaudit-Daten veröffentlicht werden dürfen.","refers_to":null,"speaker":"Antonius","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"information_question","temporality":"unspecified","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"explicit","qualifier":"Veröffentlichung von Energieaudit-Daten","limits_target":null},
{"observation_id":"obs_6","evidence_id":"e4","content":"Die Veröffentlichungserlaubnis ist weiterhin ungeklärt.","refers_to":"obs_5","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":false,"impersonal_person_reference":false,"modality":"factual","temporality":"existing","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"explicit","qualifier":"weiterhin","limits_target":null},
{"observation_id":"obs_7","evidence_id":"e4","content":"Die Freigabe muss noch geklärt werden.","refers_to":"obs_5","speaker":"Martin","named_person":null,"addressee":null,"self_reference":false,"collective_we":true,"impersonal_person_reference":false,"modality":"impersonal_necessity","temporality":"future","evaluation":"none","affirmation":"absent","negation":"absent","determination_statement":"absent","uncertainty":"present","clarification_need":"explicit","qualifier":"noch; Freigabe zur Veröffentlichung","limits_target":null}
]
}
]
}
@@ -0,0 +1,49 @@
{
"cases": [
{
"case_id":"a_idea_only","description":"Possible geometry optimization without commitment.","subject_id":"subject_a","subject":"Optimierung der Geometrie",
"evidence":[{"evidence_id":"e1","text":"Martin: Die Geometrie kann man vielleicht noch optimieren. Dann würde man mal gucken, was herauskommt."}],
"semantic_requirements":["Geometry optimization remains possible and tentative.","Subsequent checking remains conditional and tentative.","The then/sequential dependency survives.","No commitment or owner is introduced."]
},
{
"case_id":"b_multiple_options","description":"Two alternatives for insufficient grid strength.","subject_id":"subject_b","subject":"Umgang mit unzureichender Festigkeit des 40-40-Gitters",
"evidence":[{"evidence_id":"e1","text":"Martin: Die Festigkeit reicht für das 40-40-Gitter noch nicht aus."},{"evidence_id":"e2","text":"Martin: Man könnte mehr Masse für die gleiche Festigkeit einsetzen."},{"evidence_id":"e3","text":"Martin: Oder wir verkaufen es nicht als 40-40-Gitter, sondern machen ein 20-20 daraus. Das wären die zwei Ansätze."}],
"semantic_requirements":["Insufficient 40-40 strength survives.","Additional mass remains one alternative.","20-20 remains another alternative.","Both remain alternatives and neither is selected."]
},
{
"case_id":"c_unaccepted_proposal","description":"Suggested Textor contact without established work.","subject_id":"subject_c","subject":"Erneute Kontaktaufnahme mit Dirk Textor zur Einschätzung",
"evidence":[{"evidence_id":"e1","text":"Tim: Ich würde vielleicht Dirk Textor noch einmal kontaktieren und fragen, wie er das einschätzt."},{"evidence_id":"e2","text":"Tim: Das kann man ja mit ihm einfach noch einmal rückkoppeln."}],
"semantic_requirements":["Contacting Dirk Textor remains Tim's tentative personal suggestion.","The follow-up remains possible and relates to that contact.","No established work or responsibility is introduced."]
},
{
"case_id":"d_proposal_with_objection","description":"Washing possibility and explicit energy disadvantage.","subject_id":"subject_d","subject":"Waschen des Materials vor der weiteren Verarbeitung",
"evidence":[{"evidence_id":"e1","text":"Antonius: Man könnte das Material vor der weiteren Verarbeitung waschen."},{"evidence_id":"e2","text":"Martin: Ob sich das lohnt, weiß ich nicht. Waschen heißt nass machen und wieder trocknen; das ist ein wahnsinniger Energieaufwand."}],
"semantic_requirements":["Washing before further processing remains possible.","Martin's uncertainty whether washing is worthwhile survives.","The washing and drying process survives.","The high energy consequence survives.","No unresolved task is invented."]
},
{
"case_id":"e_rejected_alternative","description":"Explicit rejection followed by confirmation of that determination.","subject_id":"subject_e","subject":"Zusammenarbeit mit Dr. Schlummer für Versuche",
"evidence":[{"evidence_id":"e1","text":"Antonius: Das Angebot von Dr. Schlummer für die Versuche kostet 30.000 Euro."},{"evidence_id":"e2","text":"Tim: Dann haben wir gesagt: Nein, die Zusammenarbeit mit Dr. Schlummer machen wir nicht."},{"evidence_id":"e3","text":"Antonius: Ja, das ist entschieden."}],
"semantic_requirements":["The offer cost survives without inferred evaluation.","Collaboration is explicitly not to be pursued.","The explicit rejection survives.","The later statement confirms that the preceding determination has been made."]
},
{
"case_id":"f_trial_only_acceptance","description":"Collective commitment limited to a 20-metre trial.","subject_id":"subject_f","subject":"20-Prozent-Variante im Versuch am kleinen Extruder",
"evidence":[{"evidence_id":"e1","text":"Martin: Wir könnten die 20-Prozent-Variante am kleinen Extruder nachstellen."},{"evidence_id":"e2","text":"Tim: Ja, wir testen 20 Meter dieser Variante beim nächsten Versuch."},{"evidence_id":"e3","text":"Tim: Das ist nur ein Versuch; damit ist die Variante noch nicht als Serienlösung festgelegt."}],
"semantic_requirements":["The 20-percent variant at the small extruder remains initially possible.","The later statement collectively commits to a test.","Twenty metres and next-trial timing survive.","The test remains limited to a trial.","Series adoption remains explicitly not yet established.","No individual owner is invented."]
},
{
"case_id":"g_no_decision","description":"Preference, conditional alternative, and impersonal checking need.","subject_id":"subject_g","subject":"Reale Recyclinganlage oder Technikum und verfügbarer Reinigungsansatz",
"evidence":[{"evidence_id":"e1","text":"Martin: Eine reale Recyclinganlage hätte das Risiko, dass wir kontaminiertes Material zurückbekommen."},{"evidence_id":"e2","text":"Martin: Ich würde nicht in eine reale Anlage gehen. Wenn überhaupt, können wir über ein Technikum reden."},{"evidence_id":"e3","text":"Tim: Man müsste zunächst prüfen, welcher Reinigungsansatz überhaupt verfügbar ist."}],
"semantic_requirements":["Contamination remains a risk rather than a fact.","Martin's negative stance remains personal.","The Technikum remains conditional and if-at-all survives.","Cleaning-method availability still needs to be checked.","The need remains impersonal.","No group decision or owner is invented."]
},
{
"case_id":"h_resulting_action","description":"Interpersonal request followed by explicit personal acceptance.","subject_id":"subject_h","subject":"Prüfung der Messdaten bis Freitag",
"evidence":[{"evidence_id":"e1","text":"Antonius: Nina, übernimmst du die Prüfung der Messdaten bis Freitag?"},{"evidence_id":"e2","text":"Nina: Ja, ich übernehme die Prüfung bis Freitag."}],
"semantic_requirements":["Antonius requests measurement-data review from Nina.","The Friday deadline survives.","Nina explicitly accepts the preceding request.","Nina's response expresses future personal commitment.","No responsibility field or unsupported inference is introduced."]
},
{
"case_id":"i_outcome_and_unresolved","description":"Bounded production finding and unresolved publication information.","subject_id":"subject_i","subject":"Produktionsaufwand und Veröffentlichung von Energieaudit-Daten",
"evidence":[{"evidence_id":"e1","text":"Martin: An unserer Anlage gab es bei der reinen Produktion gegenüber dem Standardprodukt praktisch keine Änderung; wir waren nur fünf Grad kälter."},{"evidence_id":"e2","text":"Antonius: Dann können wir mindestens festhalten: Gegenüber Virgin Material ist bei der reinen Produktion kein zusätzlicher Aufwand notwendig. Davor entsteht natürlich Aufwand."},{"evidence_id":"e3","text":"Antonius: Welche Daten aus dem Energieaudit dürfen wir veröffentlichen?"},{"evidence_id":"e4","text":"Martin: Das ist weiterhin ungeklärt. Wir müssen die Freigabe noch klären."}],
"semantic_requirements":["The pure-production finding remains bounded to the local plant and standard-product comparison.","The five-degree difference survives.","No-extra-effort remains bounded to pure production compared with Virgin material.","Upstream effort before that production stage survives.","The publication purpose of the energy-audit question remains explicit.","Publication permission remains unresolved and clarification remains necessary.","No assigned work is invented."]
}
]
}
@@ -0,0 +1,11 @@
# semantic_synthesis_isolation
Isolation Gold set derived from the existing Topic Reconstruction V2 A-I
cases. Every case supplies one manually fixed Discussion Subject and the
complete original evidence bundle. The model performs Semantic Synthesis only;
subject detection, subject grouping, and evidence assignment are outside the
experiment.
Expected criteria evaluate semantic event distinctions, outcomes and scope,
actions, unresolved issues, and supporting evidence. They do not evaluate
subject discovery or exact wording.
@@ -0,0 +1,227 @@
{
"cases": [
{
"case_id": "a_idea_only",
"description": "Idea mentioned without stronger commitment.",
"subject_id": "subject_a",
"subject": "Optimierung der Geometrie",
"evidence": [
{"evidence_id": "e1", "text": "Martin: Die Geometrie kann man vielleicht noch optimieren. Dann würde man mal gucken, was herauskommt."}
],
"allowed_responsible": [],
"expected": {
"event_type_minimums": {"idea": 1},
"allowed_event_types": ["idea"],
"event_evidence_ids": ["e1"],
"outcome": {"required": false},
"actions": {"count": 0},
"unresolved_issues": {"count": 0}
}
},
{
"case_id": "b_multiple_options",
"description": "Two alternatives for insufficient 40-40 grid strength.",
"subject_id": "subject_b",
"subject": "Umgang mit unzureichender Festigkeit des 40-40-Gitters",
"evidence": [
{"evidence_id": "e1", "text": "Martin: Die Festigkeit reicht für das 40-40-Gitter noch nicht aus."},
{"evidence_id": "e2", "text": "Martin: Man könnte mehr Masse für die gleiche Festigkeit einsetzen."},
{"evidence_id": "e3", "text": "Martin: Oder wir verkaufen es nicht als 40-40-Gitter, sondern machen ein 20-20 daraus. Das wären die zwei Ansätze."}
],
"allowed_responsible": [],
"expected": {
"event_type_minimums": {"option": 2},
"allowed_event_types": ["technical_finding", "fact", "option"],
"event_evidence_ids": ["e1", "e2", "e3"],
"outcome": {"required": false},
"actions": {"count": 0},
"unresolved_issues": {"count": 0}
}
},
{
"case_id": "c_unaccepted_proposal",
"description": "Possible Textor contact remains a proposal only.",
"subject_id": "subject_c",
"subject": "Erneute Kontaktaufnahme mit Dirk Textor zur Einschätzung",
"evidence": [
{"evidence_id": "e1", "text": "Tim: Ich würde vielleicht Dirk Textor noch einmal kontaktieren und fragen, wie er das einschätzt."},
{"evidence_id": "e2", "text": "Tim: Das kann man ja mit ihm einfach noch einmal rückkoppeln."}
],
"allowed_responsible": [],
"expected": {
"event_type_minimums": {"proposal": 1},
"allowed_event_types": ["proposal"],
"event_evidence_ids": ["e1", "e2"],
"outcome": {"required": false},
"actions": {"count": 0},
"unresolved_issues": {"count": 0}
}
},
{
"case_id": "d_proposal_with_objection",
"description": "Washing proposal with energy objection but no unresolved issue.",
"subject_id": "subject_d",
"subject": "Waschen des Materials vor der weiteren Verarbeitung",
"evidence": [
{"evidence_id": "e1", "text": "Antonius: Man könnte das Material vor der weiteren Verarbeitung waschen."},
{"evidence_id": "e2", "text": "Martin: Ob sich das lohnt, weiß ich nicht. Waschen heißt nass machen und wieder trocknen; das ist ein wahnsinniger Energieaufwand."}
],
"allowed_responsible": [],
"expected": {
"event_type_minimums": {"proposal": 1, "objection": 1},
"allowed_event_types": ["proposal", "objection"],
"event_evidence_ids": ["e1", "e2"],
"outcome": {"required": false},
"actions": {"count": 0},
"unresolved_issues": {"count": 0}
}
},
{
"case_id": "e_rejected_alternative",
"description": "Explicit rejection of Schlummer collaboration.",
"subject_id": "subject_e",
"subject": "Zusammenarbeit mit Dr. Schlummer für Versuche",
"evidence": [
{"evidence_id": "e1", "text": "Antonius: Das Angebot von Dr. Schlummer für die Versuche kostet 30.000 Euro."},
{"evidence_id": "e2", "text": "Tim: Dann haben wir gesagt: Nein, die Zusammenarbeit mit Dr. Schlummer machen wir nicht."},
{"evidence_id": "e3", "text": "Antonius: Ja, das ist entschieden."}
],
"allowed_responsible": [],
"expected": {
"event_type_minimums": {"fact": 1, "rejection": 1},
"allowed_event_types": ["fact", "rejection", "clarification"],
"event_evidence_ids": ["e1", "e2", "e3"],
"outcome": {
"required": true,
"statuses": ["rejected"],
"terms": ["nicht", "abgelehnt", "keine"],
"scope_terms": ["zusammenarbeit", "versuch", "schlummer"],
"evidence_ids": ["e2", "e3"]
},
"actions": {"count": 0},
"unresolved_issues": {"count": 0}
}
},
{
"case_id": "f_trial_only_acceptance",
"description": "Acceptance limited to a 20-metre trial.",
"subject_id": "subject_f",
"subject": "20-Prozent-Variante im Versuch am kleinen Extruder",
"evidence": [
{"evidence_id": "e1", "text": "Martin: Wir könnten die 20-Prozent-Variante am kleinen Extruder nachstellen."},
{"evidence_id": "e2", "text": "Tim: Ja, wir testen 20 Meter dieser Variante beim nächsten Versuch."},
{"evidence_id": "e3", "text": "Tim: Das ist nur ein Versuch; damit ist die Variante noch nicht als Serienlösung festgelegt."}
],
"allowed_responsible": [],
"expected": {
"event_type_minimums": {"proposal": 1, "scoped_acceptance": 1, "clarification": 1},
"allowed_event_types": ["proposal", "scoped_acceptance", "clarification"],
"event_evidence_ids": ["e1", "e2", "e3"],
"outcome": {
"required": true,
"statuses": ["scoped_acceptance"],
"terms": ["test", "versuch"],
"scope_terms": ["20 meter", "20 m", "nur", "begrenzt"],
"evidence_ids": ["e2", "e3"]
},
"actions": {
"count": 1,
"terms": ["test", "versuch", "20 meter"],
"evidence_ids": ["e2"],
"due_terms": ["nächsten versuch", "next trial"]
},
"unresolved_issues": {
"count": 1,
"terms": ["serienlösung", "final", "serie", "festgelegt"],
"evidence_ids": ["e3"]
}
}
},
{
"case_id": "g_no_decision",
"description": "Plant versus Technikum discussion ending without a decision.",
"subject_id": "subject_g",
"subject": "Reale Recyclinganlage oder Technikum und verfügbarer Reinigungsansatz",
"evidence": [
{"evidence_id": "e1", "text": "Martin: Eine reale Recyclinganlage hätte das Risiko, dass wir kontaminiertes Material zurückbekommen."},
{"evidence_id": "e2", "text": "Martin: Ich würde nicht in eine reale Anlage gehen. Wenn überhaupt, können wir über ein Technikum reden."},
{"evidence_id": "e3", "text": "Tim: Man müsste zunächst prüfen, welcher Reinigungsansatz überhaupt verfügbar ist."}
],
"allowed_responsible": [],
"expected": {
"event_type_minimums": {"objection": 1, "option": 1},
"allowed_event_types": ["objection", "option", "proposal", "clarification"],
"event_evidence_ids": ["e1", "e2", "e3"],
"outcome": {"required": false},
"actions": {"count": 0},
"unresolved_issues": {
"count": 1,
"terms": ["reinigungsansatz", "verfügbar", "prüfen", "reinigung"],
"evidence_ids": ["e3"]
}
}
},
{
"case_id": "h_resulting_action",
"description": "Explicitly accepted action with owner and deadline.",
"subject_id": "subject_h",
"subject": "Prüfung der Messdaten bis Freitag",
"evidence": [
{"evidence_id": "e1", "text": "Antonius: Nina, übernimmst du die Prüfung der Messdaten bis Freitag?"},
{"evidence_id": "e2", "text": "Nina: Ja, ich übernehme die Prüfung bis Freitag."}
],
"allowed_responsible": ["Nina"],
"expected": {
"event_type_minimums": {},
"allowed_event_types": ["proposal", "clarification", "scoped_acceptance"],
"event_evidence_ids": [],
"outcome": {
"required": true,
"statuses": ["established"],
"terms": ["übernimmt", "prüf", "accepted", "review", "assigned"],
"scope_terms": ["messdaten", "prüfung", "measurement", "review"],
"evidence_ids": ["e2"]
},
"actions": {
"count": 1,
"terms": ["messdaten", "prüf", "measurement", "review"],
"responsible": "Nina",
"due_terms": ["freitag", "friday"],
"evidence_ids": ["e2"]
},
"unresolved_issues": {"count": 0}
}
},
{
"case_id": "i_outcome_and_unresolved",
"description": "Bounded production outcome and unresolved publication question.",
"subject_id": "subject_i",
"subject": "Produktionsaufwand und Veröffentlichung von Energieaudit-Daten",
"evidence": [
{"evidence_id": "e1", "text": "Martin: An unserer Anlage gab es bei der reinen Produktion gegenüber dem Standardprodukt praktisch keine Änderung; wir waren nur fünf Grad kälter."},
{"evidence_id": "e2", "text": "Antonius: Dann können wir mindestens festhalten: Gegenüber Virgin Material ist bei der reinen Produktion kein zusätzlicher Aufwand notwendig. Davor entsteht natürlich Aufwand."},
{"evidence_id": "e3", "text": "Antonius: Welche Daten aus dem Energieaudit dürfen wir veröffentlichen?"},
{"evidence_id": "e4", "text": "Martin: Das ist weiterhin ungeklärt. Wir müssen die Freigabe noch klären."}
],
"allowed_responsible": [],
"expected": {
"event_type_minimums": {"fact": 1},
"allowed_event_types": ["technical_finding", "fact", "clarification"],
"event_evidence_ids": ["e1", "e2"],
"outcome": {
"required": true,
"statuses": ["established"],
"terms": ["kein zusätzlicher", "keine zusätzliche", "unverändert"],
"scope_terms": ["reine produktion", "produktion", "virgin"],
"evidence_ids": ["e1", "e2"]
},
"actions": {"count": 0},
"unresolved_issues": {
"count": 1,
"terms": ["veröffentlich", "freigabe", "energieaudit", "daten"],
"evidence_ids": ["e3", "e4"]
}
}
}
]
}
@@ -0,0 +1,20 @@
# topic_reconstruction_v2
Focused experimental Gold material derived from BUG-015 and the Progeo
discussion. These cases evaluate topic-oriented reconstruction rather than
exact protocol wording or flat category extraction.
The nine cases cover:
- an idea mentioned without further development;
- multiple alternatives;
- an unaccepted proposal;
- a proposal with an objection;
- an explicitly rejected alternative;
- acceptance limited to a bounded trial;
- discussion ending without a decision;
- a resulting Action Item;
- an outcome accompanied by an unresolved issue.
Evidence units carry stable local IDs. Expected criteria describe semantic
features and prohibited promotions rather than exact generated sentences.
@@ -0,0 +1,258 @@
{
"cases": [
{
"case_id": "a_idea_only",
"description": "A geometry optimization idea is mentioned but not developed.",
"evidence_units": [
{
"evidence_id": "e1",
"text": "Martin: Die Geometrie kann man vielleicht noch optimieren. Dann würde man mal gucken, was herauskommt."
}
],
"expected": {
"subject_count": 1,
"subject_terms": ["geometr"],
"required_event_types": ["introduced_idea"],
"outcome": {"required": false},
"actions": {"minimum": 0},
"unresolved": {"minimum": 0}
}
},
{
"case_id": "b_multiple_options",
"description": "Two alternatives for compensating insufficient specimen strength are discussed.",
"evidence_units": [
{
"evidence_id": "e1",
"text": "Martin: Die Festigkeit reicht für das 40-40-Gitter noch nicht aus."
},
{
"evidence_id": "e2",
"text": "Martin: Man könnte mehr Masse für die gleiche Festigkeit einsetzen."
},
{
"evidence_id": "e3",
"text": "Martin: Oder wir verkaufen es nicht als 40-40-Gitter, sondern machen ein 20-20 daraus. Das wären die zwei Ansätze."
}
],
"expected": {
"subject_count": 1,
"subject_terms": ["festigkeit", "gitter", "geometr"],
"required_event_types": ["considered_option"],
"outcome": {"required": false},
"actions": {"minimum": 0},
"unresolved": {"minimum": 0}
}
},
{
"case_id": "c_unaccepted_proposal",
"description": "Contacting Dirk Textor is proposed but not accepted as work.",
"evidence_units": [
{
"evidence_id": "e1",
"text": "Tim: Ich würde vielleicht Dirk Textor noch einmal kontaktieren und fragen, wie er das einschätzt."
},
{
"evidence_id": "e2",
"text": "Tim: Das kann man ja mit ihm einfach noch einmal rückkoppeln."
}
],
"expected": {
"subject_count": 1,
"subject_terms": ["textor", "einschätzung", "kontakt"],
"required_event_types": ["proposal"],
"outcome": {"required": false},
"actions": {"minimum": 0},
"unresolved": {"minimum": 0}
}
},
{
"case_id": "d_proposal_with_objection",
"description": "Washing is considered and an energy-cost objection is raised.",
"evidence_units": [
{
"evidence_id": "e1",
"text": "Antonius: Man könnte das Material vor der weiteren Verarbeitung waschen."
},
{
"evidence_id": "e2",
"text": "Martin: Ob sich das lohnt, weiß ich nicht. Waschen heißt nass machen und wieder trocknen; das ist ein wahnsinniger Energieaufwand."
}
],
"expected": {
"subject_count": 1,
"subject_terms": ["wasch", "reinig"],
"required_event_types": ["proposal", "objection"],
"outcome": {"required": false},
"actions": {"minimum": 0},
"unresolved": {"minimum": 0}
}
},
{
"case_id": "e_rejected_alternative",
"description": "The collaboration with Dr. Schlummer is explicitly rejected after its cost is discussed.",
"evidence_units": [
{
"evidence_id": "e1",
"text": "Antonius: Das Angebot von Dr. Schlummer für die Versuche kostet 30.000 Euro."
},
{
"evidence_id": "e2",
"text": "Tim: Dann haben wir gesagt: Nein, die Zusammenarbeit mit Dr. Schlummer machen wir nicht."
},
{
"evidence_id": "e3",
"text": "Antonius: Ja, das ist entschieden."
}
],
"expected": {
"subject_count": 1,
"subject_terms": ["schlummer", "zusammenarbeit"],
"required_event_types": ["fact"],
"outcome": {
"required": true,
"terms": ["nicht", "abgelehnt", "keine"],
"scope_terms": ["zusammenarbeit", "versuch"],
"certainties": ["rejected", "established"]
},
"actions": {"minimum": 0},
"unresolved": {"minimum": 0}
}
},
{
"case_id": "f_trial_only_acceptance",
"description": "A 20 percent variant is accepted only for a bounded trial, not as the final production solution.",
"evidence_units": [
{
"evidence_id": "e1",
"text": "Martin: Wir könnten die 20-Prozent-Variante am kleinen Extruder nachstellen."
},
{
"evidence_id": "e2",
"text": "Tim: Ja, wir testen 20 Meter dieser Variante beim nächsten Versuch."
},
{
"evidence_id": "e3",
"text": "Tim: Das ist nur ein Versuch; damit ist die Variante noch nicht als Serienlösung festgelegt."
}
],
"expected": {
"subject_count": 1,
"subject_terms": ["20-prozent", "variante", "extruder"],
"required_event_types": ["proposal", "clarification"],
"outcome": {
"required": true,
"terms": ["test", "versuch"],
"scope_terms": ["20 meter", "20 m", "nur", "begrenzt"],
"certainties": ["established", "conditional"]
},
"actions": {
"minimum": 1,
"terms": ["test", "versuch", "20 meters", "20 meter"]
},
"unresolved": {
"minimum": 1,
"terms": ["final", "series", "serie", "adopt", "festgelegt"]
}
}
},
{
"case_id": "g_no_decision",
"description": "Real recycling plant and Technikum alternatives are discussed without a group decision.",
"evidence_units": [
{
"evidence_id": "e1",
"text": "Martin: Eine reale Recyclinganlage hätte das Risiko, dass wir kontaminiertes Material zurückbekommen."
},
{
"evidence_id": "e2",
"text": "Martin: Ich würde nicht in eine reale Anlage gehen. Wenn überhaupt, können wir über ein Technikum reden."
},
{
"evidence_id": "e3",
"text": "Tim: Man müsste zunächst prüfen, welcher Reinigungsansatz überhaupt verfügbar ist."
}
],
"expected": {
"subject_count": 1,
"subject_terms": ["technikum", "reinig", "anlage"],
"required_event_types": ["considered_option", "objection"],
"outcome": {"required": false},
"actions": {"minimum": 0},
"unresolved": {
"minimum": 1,
"terms": ["reinigungsansatz", "verfügbar", "anlage", "prüfen"]
}
}
},
{
"case_id": "h_resulting_action",
"description": "The discussion establishes an accepted review action with owner and deadline.",
"evidence_units": [
{
"evidence_id": "e1",
"text": "Antonius: Nina, übernimmst du die Prüfung der Messdaten bis Freitag?"
},
{
"evidence_id": "e2",
"text": "Nina: Ja, ich übernehme die Prüfung bis Freitag."
}
],
"expected": {
"subject_count": 1,
"subject_terms": ["messdaten", "prüfung", "prüfen", "verification", "data"],
"required_event_types": [],
"outcome": {
"required": true,
"terms": ["agrees", "übernimmt", "accepted", "verify"],
"scope_terms": ["measurement", "messdaten", "verification"],
"certainties": ["established"]
},
"actions": {
"minimum": 1,
"terms": ["messdaten", "prüf", "verify", "measurement"],
"responsible": "Nina"
},
"unresolved": {"minimum": 0}
}
},
{
"case_id": "i_outcome_and_unresolved",
"description": "The production-energy discussion establishes one bounded finding while publication remains unresolved.",
"evidence_units": [
{
"evidence_id": "e1",
"text": "Martin: An unserer Anlage gab es bei der reinen Produktion gegenüber dem Standardprodukt praktisch keine Änderung; wir waren nur fünf Grad kälter."
},
{
"evidence_id": "e2",
"text": "Antonius: Dann können wir mindestens festhalten: Gegenüber Virgin Material ist bei der reinen Produktion kein zusätzlicher Aufwand notwendig. Davor entsteht natürlich Aufwand."
},
{
"evidence_id": "e3",
"text": "Antonius: Welche Daten aus dem Energieaudit dürfen wir veröffentlichen?"
},
{
"evidence_id": "e4",
"text": "Martin: Das ist weiterhin ungeklärt. Wir müssen die Freigabe noch klären."
}
],
"expected": {
"subject_count": 1,
"subject_terms": ["energie", "aufwand", "produktion"],
"required_event_types": ["technical_finding"],
"outcome": {
"required": true,
"terms": ["kein zusätzlicher", "keine zusätzliche", "unverändert"],
"scope_terms": ["reine produktion", "produktion", "gegenüber virgin"],
"certainties": ["established"]
},
"actions": {"minimum": 0},
"unresolved": {
"minimum": 1,
"terms": ["veröffentlich", "freigabe", "energieaudit", "daten"]
}
}
}
]
}
@@ -0,0 +1,168 @@
import json
import tempfile
import unittest
from copy import deepcopy
from pathlib import Path
from unittest.mock import patch
from src.meeting_lab.evidence_observations.experiment import (
SCHEMA_VERSION,
ObservationValidationError,
build_ollama_payload,
load_fixture,
parse_model_json,
run_case,
validate_observations,
)
class EvidenceObservationExperimentTests(unittest.TestCase):
def setUp(self) -> None:
self.case = {
"case_id": "test_case",
"description": "Validator fixture.",
"subject_id": "subject_test",
"subject": "Prüfung der Messdaten",
"evidence": [
{"evidence_id": "e1", "text": "Nina, prüfst du die Daten?"},
{"evidence_id": "e2", "text": "Ja, ich prüfe sie."},
],
"expected_observations": [],
}
self.output = {
"schema_version": SCHEMA_VERSION,
"subject_id": "subject_test",
"subject": "Prüfung der Messdaten",
"observations": [self.observation()],
}
self.case["expected_observations"] = deepcopy(self.output["observations"])
def observation(self, **updates):
value = {
"observation_id": "obs_1",
"evidence_id": "e1",
"content": "Nina wird um Prüfung gebeten.",
"target": "discussion_subject",
"relation": "none",
"modality": "interpersonal_request",
"temporality": "future",
"evaluation": "none",
"agreement": "none",
"responsibility": "named",
"person": "Nina",
"uncertainty": "absent",
"clarification_need": "none",
"scope": "absent",
}
value.update(updates)
return value
def test_valid_observation_and_discussion_subject_target(self):
self.assertIs(validate_observations(self.output, self.case), self.output)
def test_multiple_observations_from_one_evidence_unit_and_observation_target(self):
second = self.observation(
observation_id="obs_2", target="obs_1", relation="supports"
)
self.output["observations"].append(second)
validate_observations(self.output, self.case)
def test_plural_target_is_allowed_for_joint_reference(self):
self.output["observations"].extend(
[
self.observation(observation_id="obs_2"),
self.observation(
observation_id="obs_3",
target=["obs_1", "obs_2"],
relation="qualifies",
),
]
)
validate_observations(self.output, self.case)
def test_unknown_evidence_reference_is_rejected(self):
self.output["observations"][0]["evidence_id"] = "missing"
with self.assertRaisesRegex(ObservationValidationError, "unknown evidence"):
validate_observations(self.output, self.case)
def test_unknown_observation_target_is_rejected(self):
self.output["observations"][0]["target"] = "obs_9"
with self.assertRaisesRegex(ObservationValidationError, "unknown or later"):
validate_observations(self.output, self.case)
def test_invalid_relation_is_rejected(self):
self.output["observations"][0]["relation"] = "causes"
with self.assertRaisesRegex(ObservationValidationError, "relation is invalid"):
validate_observations(self.output, self.case)
def test_invalid_modality_is_rejected(self):
self.output["observations"][0]["modality"] = "proposal"
with self.assertRaisesRegex(ObservationValidationError, "modality is invalid"):
validate_observations(self.output, self.case)
def test_invalid_responsibility_person_combinations_are_rejected(self):
self.output["observations"][0].update(responsibility="none", person="Nina")
with self.assertRaisesRegex(ObservationValidationError, "person must be JSON null"):
validate_observations(self.output, self.case)
self.output["observations"][0].update(responsibility="accepted", person=None)
with self.assertRaisesRegex(ObservationValidationError, "person must be a non-empty"):
validate_observations(self.output, self.case)
def test_scope_uses_absent_or_nonempty_evidence_grounded_text(self):
validate_observations(self.output, self.case)
self.output["observations"][0]["scope"] = "bis Freitag"
validate_observations(self.output, self.case)
self.output["observations"][0]["scope"] = None
with self.assertRaisesRegex(ObservationValidationError, "non-empty string"):
validate_observations(self.output, self.case)
def test_string_null_is_rejected_in_text_fields(self):
self.output["observations"][0]["scope"] = "null"
with self.assertRaisesRegex(ObservationValidationError, "string 'null'"):
validate_observations(self.output, self.case)
def test_malformed_model_json_is_rejected(self):
with self.assertRaises(json.JSONDecodeError):
parse_model_json("{not json")
def test_payload_has_exact_live_controls(self):
payload = build_ollama_payload("qwen3.5:9B", "prompt", 16384, 4096)
self.assertIs(payload["think"], False)
self.assertIs(payload["stream"], False)
self.assertEqual(payload["format"], "json")
self.assertEqual(payload["options"]["temperature"], 0)
def test_fixture_contains_all_nine_cases(self):
cases = load_fixture(Path("tests/gold/evidence_observations_v1/cases.json"))
self.assertEqual(len(cases), 9)
self.assertEqual(cases[0]["case_id"], "a_idea_only")
self.assertEqual(cases[-1]["case_id"], "i_outcome_and_unresolved")
def test_case_run_preserves_all_artifacts(self):
raw = json.dumps(self.output, ensure_ascii=False)
with tempfile.TemporaryDirectory() as temporary:
root = Path(temporary)
with patch(
"src.meeting_lab.evidence_observations.experiment.call_ollama",
return_value=(raw, {"model": "qwen3.5:9B"}),
):
result = run_case(
self.case, root, "http://unused", "qwen3.5:9B", 1, 16384, 4096
)
self.assertEqual(result["verdict"], "PASS")
for filename in (
"gold_input.json",
"gold_expected_observations.json",
"prompt.txt",
"raw_model_response.txt",
"parsed_observations.json",
"validation_result.json",
"ollama_metadata.json",
"evaluation_result.json",
):
self.assertTrue((root / "test_case" / filename).is_file(), filename)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,160 @@
import json
import tempfile
import unittest
from copy import deepcopy
from pathlib import Path
from unittest.mock import patch
from src.meeting_lab.evidence_observations_v2.experiment import (
SCHEMA_VERSION,
ObservationValidationError,
build_ollama_payload,
load_fixture,
parse_model_json,
run_case,
validate_observations,
)
class EvidenceObservationV2ExperimentTests(unittest.TestCase):
def setUp(self) -> None:
self.case = {
"case_id": "test_case", "description": "Validator fixture.",
"subject_id": "subject_test", "subject": "Prüfung der Messdaten",
"evidence": [
{"evidence_id": "e1", "text": "Antonius: Nina, prüfst du die Daten?"},
{"evidence_id": "e2", "text": "Nina: Ja, ich prüfe sie."},
],
"expected_observations": [],
}
self.output = {
"schema_version": SCHEMA_VERSION,
"subject_id": self.case["subject_id"], "subject": self.case["subject"],
"observations": [self.observation()],
}
self.case["expected_observations"] = deepcopy(self.output["observations"])
def observation(self, **updates):
value = {
"observation_id": "obs_1", "evidence_id": "e1",
"content": "Antonius bittet Nina um eine Prüfung.", "refers_to": None,
"speaker": "Antonius", "named_person": "Nina", "addressee": "Nina",
"self_reference": False, "collective_we": False,
"impersonal_person_reference": False,
"modality": "interpersonal_request", "temporality": "future",
"evaluation": "none", "affirmation": "absent", "negation": "absent",
"determination_statement": "absent", "uncertainty": "absent",
"clarification_need": "none", "qualifier": "bis Freitag",
"limits_target": None,
}
value.update(updates)
return value
def test_participant_facts_do_not_include_responsibility(self):
validate_observations(self.output, self.case)
observation = self.output["observations"][0]
self.assertEqual(observation["speaker"], "Antonius")
self.assertEqual(observation["named_person"], "Nina")
self.assertEqual(observation["addressee"], "Nina")
self.assertNotIn("responsibility", observation)
def test_named_person_and_speaker_do_not_imply_any_extra_field(self):
keys = self.output["observations"][0].keys()
self.assertNotIn("person", keys)
self.assertNotIn("agreement", keys)
def test_self_reference_collective_we_and_impersonal_reference_are_boolean(self):
self.output["observations"][0].update(
self_reference=True, collective_we=True, impersonal_person_reference=True
)
validate_observations(self.output, self.case)
self.output["observations"][0]["collective_we"] = "true"
with self.assertRaisesRegex(ObservationValidationError, "must be boolean"):
validate_observations(self.output, self.case)
def test_explicit_affirmation_negation_and_determination(self):
self.output["observations"][0].update(
affirmation="explicit", negation="explicit", determination_statement="present"
)
validate_observations(self.output, self.case)
def test_scalar_reference_to_prior_observation(self):
self.output["observations"].append(self.observation(
observation_id="obs_2", evidence_id="e2", refers_to="obs_1",
speaker="Nina", named_person=None, addressee=None,
))
validate_observations(self.output, self.case)
def test_array_and_invalid_reference_are_rejected(self):
self.output["observations"][0]["refers_to"] = ["obs_1"]
with self.assertRaisesRegex(ObservationValidationError, "non-empty string"):
validate_observations(self.output, self.case)
self.output["observations"][0]["refers_to"] = "obs_9"
with self.assertRaisesRegex(ObservationValidationError, "unknown or later"):
validate_observations(self.output, self.case)
def test_qualifier_is_null_or_nonempty_text(self):
self.output["observations"][0]["qualifier"] = None
validate_observations(self.output, self.case)
self.output["observations"][0]["qualifier"] = ""
with self.assertRaisesRegex(ObservationValidationError, "non-empty string"):
validate_observations(self.output, self.case)
def test_limits_target_must_reference_prior_observation(self):
self.output["observations"].append(self.observation(
observation_id="obs_2", evidence_id="e2", refers_to="obs_1",
limits_target="obs_1", speaker="Nina", named_person=None, addressee=None,
))
validate_observations(self.output, self.case)
self.output["observations"][1]["limits_target"] = "obs_7"
with self.assertRaisesRegex(ObservationValidationError, "unknown or later"):
validate_observations(self.output, self.case)
def test_multiple_atomic_observations_may_share_evidence(self):
self.output["observations"].append(self.observation(observation_id="obs_2"))
validate_observations(self.output, self.case)
def test_string_null_is_rejected(self):
self.output["observations"][0]["named_person"] = "null"
with self.assertRaisesRegex(ObservationValidationError, "string 'null'"):
validate_observations(self.output, self.case)
def test_malformed_json_is_rejected(self):
with self.assertRaises(json.JSONDecodeError):
parse_model_json("{not json")
def test_payload_has_exact_live_controls(self):
payload = build_ollama_payload("qwen3.5:9B", "prompt", 16384, 4096)
self.assertFalse(payload["think"])
self.assertFalse(payload["stream"])
self.assertEqual(payload["options"]["temperature"], 0)
def test_fixture_contains_unchanged_a_i_source_evidence(self):
v1 = load_fixture(Path("tests/gold/evidence_observations_v2/cases.json"))
original = json.loads(Path("tests/gold/evidence_observations_v1/cases.json").read_text())["cases"]
self.assertEqual(len(v1), 9)
self.assertEqual(
[[item["text"] for item in case["evidence"]] for case in v1],
[[item["text"] for item in case["evidence"]] for case in original],
)
def test_case_run_preserves_all_artifacts(self):
raw = json.dumps(self.output, ensure_ascii=False)
with tempfile.TemporaryDirectory() as temporary:
root = Path(temporary)
with patch(
"src.meeting_lab.evidence_observations_v2.experiment.call_ollama",
return_value=(raw, {"model": "qwen3.5:9B"}),
):
result = run_case(self.case, root, "http://unused", "qwen3.5:9B", 1, 16384, 4096)
self.assertEqual(result["verdict"], "PASS")
for filename in (
"gold_input.json", "gold_expected_observations.json", "prompt.txt",
"raw_model_response.txt", "parsed_observations.json",
"validation_result.json", "ollama_metadata.json", "evaluation_result.json",
):
self.assertTrue((root / "test_case" / filename).is_file(), filename)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,114 @@
import json
import tempfile
import unittest
from pathlib import Path
from unittest.mock import patch
from src.meeting_lab.evidence_observations_v3.experiment import (
SCHEMA_VERSION,
ObservationValidationError,
build_ollama_payload,
load_fixture,
parse_model_json,
run_case,
validate_observations,
)
class EvidenceObservationV3ExperimentTests(unittest.TestCase):
def setUp(self) -> None:
self.case = {
"case_id": "test", "description": "Minimal fixture.",
"subject_id": "subject_test", "subject": "Messdatenprüfung",
"evidence": [
{"evidence_id": "e1", "text": "Antonius: Nina, prüfst du die Messdaten?"},
{"evidence_id": "e2", "text": "Nina: Ja, ich prüfe sie bis Freitag."},
],
"semantic_requirements": ["Request and response survive."],
}
self.output = {
"schema_version": SCHEMA_VERSION, "subject_id": "subject_test",
"subject": "Messdatenprüfung", "observations": [self.observation()],
}
def observation(self, **updates):
value = {
"observation_id": "obs_1", "evidence_id": "e1",
"content": "Antonius fragt Nina, ob sie die Messdaten prüft.",
"speaker": "Antonius", "named_person": "Nina", "addressee": "Nina",
}
value.update(updates)
return value
def test_minimal_schema_is_valid(self):
self.assertIs(validate_observations(self.output, self.case), self.output)
self.assertEqual(set(self.output["observations"][0]), {
"observation_id", "evidence_id", "content", "speaker", "named_person", "addressee"
})
def test_unknown_semantic_field_is_rejected(self):
self.output["observations"][0]["modality"] = "factual"
with self.assertRaisesRegex(ObservationValidationError, "unknown keys"):
validate_observations(self.output, self.case)
def test_unknown_evidence_is_rejected(self):
self.output["observations"][0]["evidence_id"] = "e9"
with self.assertRaisesRegex(ObservationValidationError, "unknown evidence"):
validate_observations(self.output, self.case)
def test_speaker_must_match_evidence(self):
self.output["observations"][0]["speaker"] = "Nina"
with self.assertRaisesRegex(ObservationValidationError, "match evidence speaker"):
validate_observations(self.output, self.case)
def test_named_person_does_not_add_responsibility(self):
validate_observations(self.output, self.case)
self.assertNotIn("responsibility", self.output["observations"][0])
def test_addressee_does_not_add_assignment(self):
validate_observations(self.output, self.case)
self.assertNotIn("action_item", self.output["observations"][0])
def test_nonexplicit_person_is_rejected(self):
self.output["observations"][0]["named_person"] = "Martin"
with self.assertRaisesRegex(ObservationValidationError, "not an explicit person"):
validate_observations(self.output, self.case)
def test_multiple_atomic_observations_can_share_evidence(self):
self.output["observations"].append(self.observation(observation_id="obs_2"))
validate_observations(self.output, self.case)
def test_malformed_json_is_rejected(self):
with self.assertRaises(json.JSONDecodeError):
parse_model_json("{bad json")
def test_payload_controls_are_fixed(self):
payload = build_ollama_payload("qwen3.5:9B", "prompt", 16384, 4096)
self.assertFalse(payload["think"])
self.assertEqual(payload["options"]["temperature"], 0)
def test_fixture_reuses_exact_v2_evidence(self):
v3 = load_fixture(Path("tests/gold/evidence_observations_v3/cases.json"))
v2 = json.loads(Path("tests/gold/evidence_observations_v2/cases.json").read_text())["cases"]
self.assertEqual([case["evidence"] for case in v3], [case["evidence"] for case in v2])
def test_case_run_preserves_persistent_artifact_set(self):
raw = json.dumps(self.output, ensure_ascii=False)
with tempfile.TemporaryDirectory() as temporary:
root = Path(temporary)
with patch(
"src.meeting_lab.evidence_observations_v3.experiment.call_ollama",
return_value=(raw, {"model": "qwen3.5:9B"}),
):
result = run_case(self.case, root, "http://unused", "qwen3.5:9B", 1, 16384, 4096)
self.assertTrue(result["structurally_valid"])
for filename in (
"source_evidence.json", "gold_semantic_requirements.json", "prompt.txt",
"raw_model_response.txt", "parsed_observations.json",
"structural_validation.json", "ollama_metadata.json",
):
self.assertTrue((root / "test" / filename).is_file(), filename)
if __name__ == "__main__":
unittest.main()
+250
View File
@@ -0,0 +1,250 @@
import json
import tempfile
import unittest
from pathlib import Path
from unittest.mock import patch
from src.meeting_lab.semantic_synthesis.experiment import (
SCHEMA_VERSION,
SynthesisValidationError,
build_ollama_payload,
evaluate_synthesis,
load_fixture,
run_case,
validate_bundle,
validate_synthesis,
)
class SemanticSynthesisExperimentTests(unittest.TestCase):
def setUp(self) -> None:
self.case = {
"case_id": "case_1",
"description": "Known subject test.",
"subject_id": "subject_1",
"subject": "Prüfung der Messdaten",
"evidence": [
{"evidence_id": "e1", "text": "Nina übernimmt die Prüfung."},
{"evidence_id": "e2", "text": "Die Freigabe bleibt offen."},
],
"allowed_responsible": ["Nina"],
"expected": {
"event_type_minimums": {"proposal": 1},
"allowed_event_types": ["proposal"],
"event_evidence_ids": ["e1"],
"outcome": {
"required": True,
"statuses": ["established"],
"terms": ["prüfung"],
"scope_terms": ["messdaten"],
"evidence_ids": ["e1"],
},
"actions": {
"count": 1,
"terms": ["prüfung"],
"responsible": "Nina",
"due_terms": [],
"evidence_ids": ["e1"],
},
"unresolved_issues": {
"count": 1,
"terms": ["freigabe"],
"evidence_ids": ["e2"],
},
},
}
def valid_output(self):
return {
"schema_version": SCHEMA_VERSION,
"subject_id": "subject_1",
"subject": "Prüfung der Messdaten",
"events": [
{
"type": "proposal",
"text": "Die Prüfung wird vorgeschlagen.",
"evidence_ids": ["e1"],
}
],
"outcome": {
"status": "established",
"text": "Die Prüfung wird übernommen.",
"scope": "Prüfung der Messdaten",
"evidence_ids": ["e1"],
},
"actions": [
{
"text": "Prüfung der Messdaten durchführen.",
"responsible": "Nina",
"due": None,
"evidence_ids": ["e1"],
}
],
"unresolved_issues": [
{
"text": "Die Freigabe bleibt offen.",
"evidence_ids": ["e2"],
}
],
}
def test_bundle_validation_accepts_fixed_subject_and_complete_evidence(self):
self.assertIs(validate_bundle(self.case), self.case)
def test_bundle_validation_rejects_duplicate_evidence_ids(self):
case = dict(self.case)
case["evidence"] = self.case["evidence"] * 2
with self.assertRaisesRegex(SynthesisValidationError, "duplicate evidence ID"):
validate_bundle(case)
def test_sparse_absence_uses_empty_arrays_and_omitted_outcome(self):
output = {
"schema_version": SCHEMA_VERSION,
"subject_id": "subject_1",
"subject": "Prüfung der Messdaten",
"events": [],
"actions": [],
"unresolved_issues": [],
}
self.assertIs(validate_synthesis(output, self.case), output)
def test_outcome_null_is_rejected_but_omission_is_allowed(self):
output = self.valid_output()
output["outcome"] = None
with self.assertRaisesRegex(SynthesisValidationError, "omit it when absent"):
validate_synthesis(output, self.case)
def test_required_arrays_must_exist(self):
for field in ("events", "actions", "unresolved_issues"):
with self.subTest(field=field):
output = self.valid_output()
del output[field]
with self.assertRaisesRegex(SynthesisValidationError, "missing required"):
validate_synthesis(output, self.case)
def test_fixed_subject_identity_cannot_change(self):
output = self.valid_output()
output["subject"] = "Different subject"
with self.assertRaisesRegex(SynthesisValidationError, "changed fixed subject"):
validate_synthesis(output, self.case)
def test_unknown_evidence_id_is_rejected_in_every_structure(self):
mutations = (
lambda output: output["events"][0].update(evidence_ids=["unknown"]),
lambda output: output["outcome"].update(evidence_ids=["unknown"]),
lambda output: output["actions"][0].update(evidence_ids=["unknown"]),
lambda output: output["unresolved_issues"][0].update(
evidence_ids=["unknown"]
),
)
for mutate in mutations:
output = self.valid_output()
mutate(output)
with self.assertRaisesRegex(SynthesisValidationError, "unknown evidence ID"):
validate_synthesis(output, self.case)
def test_duplicate_evidence_reference_is_rejected(self):
output = self.valid_output()
output["events"][0]["evidence_ids"] = ["e1", "e1"]
with self.assertRaisesRegex(SynthesisValidationError, "duplicate evidence ID"):
validate_synthesis(output, self.case)
def test_responsibility_must_be_allowed_or_json_null(self):
output = self.valid_output()
output["actions"][0]["responsible"] = None
validate_synthesis(output, self.case)
output["actions"][0]["responsible"] = "Martin"
with self.assertRaisesRegex(SynthesisValidationError, "not allowed"):
validate_synthesis(output, self.case)
def test_string_null_is_rejected(self):
output = self.valid_output()
output["actions"][0]["due"] = "null"
with self.assertRaisesRegex(SynthesisValidationError, "JSON null"):
validate_synthesis(output, self.case)
def test_outcome_action_and_unresolved_structures_are_strict(self):
for field, target in (
("extra", lambda output: output["outcome"]),
("extra", lambda output: output["actions"][0]),
("extra", lambda output: output["unresolved_issues"][0]),
):
output = self.valid_output()
target(output)[field] = "not allowed"
with self.assertRaisesRegex(SynthesisValidationError, "unknown keys"):
validate_synthesis(output, self.case)
def test_evaluator_passes_complete_semantics(self):
result = evaluate_synthesis(self.valid_output(), self.case["expected"])
self.assertEqual(result["verdict"], "PASS")
def test_evaluator_treats_invented_action_as_critical(self):
output = self.valid_output()
expected = dict(self.case["expected"])
expected["actions"] = {"count": 0}
result = evaluate_synthesis(output, expected)
self.assertEqual(result["verdict"], "FAIL")
self.assertIn("action_count", result["critical_failures"])
def test_ollama_payload_is_bounded_and_has_required_controls(self):
payload = build_ollama_payload("qwen3.5:9B", "prompt", 8192, 2048)
self.assertEqual(payload["format"], "json")
self.assertIs(payload["think"], False)
self.assertIs(payload["stream"], False)
self.assertEqual(payload["options"]["temperature"], 0)
self.assertEqual(payload["options"]["num_ctx"], 8192)
self.assertEqual(payload["options"]["num_predict"], 2048)
def test_fixture_contains_all_nine_isolation_cases(self):
cases = load_fixture(Path("tests/gold/semantic_synthesis_isolation/cases.json"))
self.assertEqual(
[case["case_id"] for case in cases],
[
"a_idea_only",
"b_multiple_options",
"c_unaccepted_proposal",
"d_proposal_with_objection",
"e_rejected_alternative",
"f_trial_only_acceptance",
"g_no_decision",
"h_resulting_action",
"i_outcome_and_unresolved",
],
)
def test_case_run_preserves_all_inspection_artifacts(self):
raw = json.dumps(self.valid_output(), ensure_ascii=False)
metadata = {"elapsed_seconds": 0.01}
with tempfile.TemporaryDirectory() as temporary:
root = Path(temporary)
with patch(
"src.meeting_lab.semantic_synthesis.experiment.call_ollama",
return_value=(raw, metadata),
):
result = run_case(
self.case,
root,
"http://unused",
"qwen3.5:9B",
1,
8192,
2048,
)
self.assertEqual(result["verdict"], "PASS")
case_dir = root / "case_1"
for filename in (
"gold_input.json",
"prompt.txt",
"raw_model_response.txt",
"parsed_response.json",
"ollama_metadata.json",
"validation_result.json",
"evaluation_result.json",
):
self.assertTrue((case_dir / filename).exists(), filename)
if __name__ == "__main__":
unittest.main()
@@ -0,0 +1,292 @@
import json
import tempfile
import unittest
from pathlib import Path
from unittest.mock import patch
from src.meeting_lab.topic_reconstruction.experiment import (
ReconstructionValidationError,
SCHEMA_VERSION,
build_ollama_payload,
evaluate_reconstruction,
run_case,
validate_evidence_units,
validate_reconstruction,
)
class TopicReconstructionExperimentTests(unittest.TestCase):
def setUp(self) -> None:
self.evidence = [
{"evidence_id": "e1", "text": "Eine Variante wird vorgeschlagen."},
{"evidence_id": "e2", "text": "Die Variante wird nur getestet."},
{"evidence_id": "e3", "text": "Nina übernimmt die Prüfung."},
{"evidence_id": "e4", "text": "Die Freigabe bleibt ungeklärt."},
]
def valid_output(self):
return {
"schema_version": SCHEMA_VERSION,
"subjects": [
{
"subject_id": "subject_1",
"title": "Versuch mit der Variante",
"evidence_refs": ["e1", "e2", "e3", "e4"],
"development": [
{
"event_id": "event_1",
"type": "proposal",
"text": "Die Variante wurde für einen Versuch vorgeschlagen.",
"evidence_refs": ["e1"],
}
],
"outcome": {
"text": "Die Variante wird getestet.",
"scope": "Nur für den Versuch, nicht als endgültige Lösung.",
"certainty": "established",
"evidence_refs": ["e2"],
},
"actions": [
{
"action_id": "action_1",
"text": "Die Variante prüfen.",
"responsible": "Nina",
"deadline": None,
"evidence_refs": ["e3"],
}
],
"unresolved_issues": [
{
"issue_id": "issue_1",
"text": "Die Freigabe ist ungeklärt.",
"evidence_refs": ["e4"],
}
],
}
],
}
def test_schema_validation_accepts_sparse_subject(self):
output = {
"schema_version": SCHEMA_VERSION,
"subjects": [
{
"subject_id": "subject_1",
"title": "Geometrie",
"evidence_refs": ["e1"],
}
],
}
self.assertIs(validate_reconstruction(output, self.evidence), output)
def test_schema_validation_accepts_complete_structures(self):
output = self.valid_output()
self.assertIs(validate_reconstruction(output, self.evidence), output)
def test_every_semantic_structure_requires_evidence_traceability(self):
structures = [
("subject", lambda data: data["subjects"][0].update(evidence_refs=[])),
(
"event",
lambda data: data["subjects"][0]["development"][0].update(
evidence_refs=[]
),
),
(
"outcome",
lambda data: data["subjects"][0]["outcome"].update(evidence_refs=[]),
),
(
"action",
lambda data: data["subjects"][0]["actions"][0].update(
evidence_refs=[]
),
),
(
"unresolved",
lambda data: data["subjects"][0]["unresolved_issues"][0].update(
evidence_refs=[]
),
),
]
for name, mutate in structures:
with self.subTest(name=name):
data = self.valid_output()
mutate(data)
with self.assertRaisesRegex(
ReconstructionValidationError, "non-empty list"
):
validate_reconstruction(data, self.evidence)
def test_unknown_evidence_reference_is_rejected(self):
output = self.valid_output()
output["subjects"][0]["outcome"]["evidence_refs"] = ["e999"]
with self.assertRaisesRegex(
ReconstructionValidationError, "unknown evidence ID: e999"
):
validate_reconstruction(output, self.evidence)
def test_duplicate_semantic_identifier_is_rejected(self):
output = self.valid_output()
output["subjects"][0]["actions"][0]["action_id"] = "event_1"
with self.assertRaisesRegex(
ReconstructionValidationError, "duplicate identifier: event_1"
):
validate_reconstruction(output, self.evidence)
def test_duplicate_input_evidence_identifier_is_rejected(self):
evidence = self.evidence + [
{"evidence_id": "e1", "text": "Duplicate source."}
]
with self.assertRaisesRegex(
ReconstructionValidationError, "duplicate input evidence identifier"
):
validate_evidence_units(evidence)
def test_empty_subjects_are_rejected(self):
output = {"schema_version": SCHEMA_VERSION, "subjects": []}
with self.assertRaisesRegex(
ReconstructionValidationError, "subjects must be a non-empty list"
):
validate_reconstruction(output, self.evidence)
def test_blank_subject_title_is_rejected(self):
output = self.valid_output()
output["subjects"][0]["title"] = " "
with self.assertRaisesRegex(
ReconstructionValidationError, "title must be a non-empty string"
):
validate_reconstruction(output, self.evidence)
def test_empty_optional_structures_must_be_omitted(self):
for field, value in (
("development", []),
("outcome", None),
("actions", []),
("unresolved_issues", []),
):
with self.subTest(field=field):
output = {
"schema_version": SCHEMA_VERSION,
"subjects": [
{
"subject_id": "subject_1",
"title": "Subject",
"evidence_refs": ["e1"],
field: value,
}
],
}
with self.assertRaises(ReconstructionValidationError):
validate_reconstruction(output, self.evidence)
def test_outcome_requires_scope_and_valid_certainty(self):
output = self.valid_output()
output["subjects"][0]["outcome"]["scope"] = ""
with self.assertRaisesRegex(ReconstructionValidationError, "scope"):
validate_reconstruction(output, self.evidence)
output = self.valid_output()
output["subjects"][0]["outcome"]["certainty"] = "accepted_forever"
with self.assertRaisesRegex(ReconstructionValidationError, "certainty"):
validate_reconstruction(output, self.evidence)
def test_action_nullable_fields_and_unresolved_structure_are_strict(self):
output = self.valid_output()
output["subjects"][0]["actions"][0]["responsible"] = None
validate_reconstruction(output, self.evidence)
output["subjects"][0]["unresolved_issues"][0]["extra"] = "invented"
with self.assertRaisesRegex(ReconstructionValidationError, "unknown keys"):
validate_reconstruction(output, self.evidence)
def test_action_nullable_fields_reject_string_null(self):
output = self.valid_output()
output["subjects"][0]["actions"][0]["responsible"] = "null"
with self.assertRaisesRegex(ReconstructionValidationError, "JSON null"):
validate_reconstruction(output, self.evidence)
def test_ollama_payload_is_bounded_and_disables_thinking(self):
payload = build_ollama_payload("qwen3.5:9B", "prompt", 16384, 4096)
self.assertEqual(payload["model"], "qwen3.5:9B")
self.assertEqual(payload["format"], "json")
self.assertIs(payload["stream"], False)
self.assertIs(payload["think"], False)
self.assertEqual(payload["options"]["temperature"], 0)
self.assertEqual(payload["options"]["num_ctx"], 16384)
self.assertEqual(payload["options"]["num_predict"], 4096)
def test_evaluator_marks_invented_action_as_critical_failure(self):
output = self.valid_output()
expected = {
"subject_count": 1,
"subject_terms": ["variante"],
"required_event_types": ["proposal"],
"outcome": {
"required": True,
"terms": ["getestet"],
"scope_terms": ["nur"],
"certainties": ["established"],
},
"actions": {"minimum": 0},
"unresolved": {"minimum": 1, "terms": ["freigabe"]},
}
result = evaluate_reconstruction(output, expected)
self.assertEqual(result["verdict"], "FAIL")
self.assertIn("action_count", result["critical_failures"])
def test_validation_failure_preserves_inspection_artifacts(self):
invalid = self.valid_output()
invalid["subjects"][0]["outcome"]["evidence_refs"] = ["unknown"]
raw = json.dumps(invalid, ensure_ascii=False)
case = {
"case_id": "artifact_case",
"description": "Artifact preservation test.",
"evidence_units": self.evidence,
"expected": {},
}
metadata = {"elapsed_seconds": 0.01}
with tempfile.TemporaryDirectory() as temporary:
root = Path(temporary)
with patch(
"src.meeting_lab.topic_reconstruction.experiment.call_ollama",
return_value=(raw, metadata),
):
result = run_case(
case,
root,
"http://unused",
"qwen3.5:9B",
1,
1024,
256,
)
case_dir = root / "artifact_case"
self.assertEqual(result["verdict"], "FAIL")
self.assertIn("schema_validation", result["critical_failures"])
self.assertTrue((case_dir / "input.json").exists())
self.assertTrue((case_dir / "prompt.txt").exists())
self.assertTrue((case_dir / "raw_model_response.txt").exists())
self.assertTrue((case_dir / "parsed_output.json").exists())
self.assertTrue((case_dir / "ollama_metadata.json").exists())
failure = json.loads(
(case_dir / "validation_failure.json").read_text(encoding="utf-8")
)
self.assertEqual(failure["error_type"], "ReconstructionValidationError")
if __name__ == "__main__":
unittest.main()