From fd7d5e1424b5683e9a123f98847fae4817335a86 Mon Sep 17 00:00:00 2001 From: Martin Date: Sun, 9 Aug 2026 16:15:35 +0200 Subject: [PATCH] Improve semantic classification precision for BUG-015 --- docs/architecture.md | 36 ++++++ docs/experiments.md | 121 +++++++++++++++++- docs/regression-bugs.md | 78 +++++++++++ prompts/classification.md | 55 ++++++++ prompts/common.md | 17 ++- prompts/questions.md | 0 src/meeting_lab/extraction/extract_chunks.py | 2 +- src/meeting_lab/llm/prompts.py | 2 +- tests/gold/progeo_action_precision/README.md | 8 ++ .../progeo_action_precision/expected.json | 21 +++ .../progeo_action_precision/transcript.txt | 11 ++ .../gold/progeo_decision_precision/README.md | 6 + .../progeo_decision_precision/expected.json | 13 ++ .../progeo_decision_precision/transcript.txt | 9 ++ .../gold/progeo_question_precision/README.md | 8 ++ .../progeo_question_precision/expected.json | 13 ++ .../progeo_question_precision/transcript.txt | 11 ++ tests/test_classification_invariants.py | 36 ++++++ tests/test_extraction_protocol.py | 32 ++++- 19 files changed, 464 insertions(+), 15 deletions(-) create mode 100644 prompts/classification.md delete mode 100644 prompts/questions.md create mode 100644 tests/gold/progeo_action_precision/README.md create mode 100644 tests/gold/progeo_action_precision/expected.json create mode 100644 tests/gold/progeo_action_precision/transcript.txt create mode 100644 tests/gold/progeo_decision_precision/README.md create mode 100644 tests/gold/progeo_decision_precision/expected.json create mode 100644 tests/gold/progeo_decision_precision/transcript.txt create mode 100644 tests/gold/progeo_question_precision/README.md create mode 100644 tests/gold/progeo_question_precision/expected.json create mode 100644 tests/gold/progeo_question_precision/transcript.txt create mode 100644 tests/test_classification_invariants.py diff --git a/docs/architecture.md b/docs/architecture.md index 343d946..08ed154 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -375,6 +375,20 @@ extraction results into `meeting_protocol.md` for technical validation. The planned architecture separates Canonical Meeting Knowledge from the final Output Views documented in `output-views.md`. +## Extraction classification contract + +Decision, Action Item and Open Question extraction shares one evidence-oriented +classification contract. It is placed after the transcript so it remains the +final classification instruction in the single multi-category extraction call. +Decisions require a settled outcome; Action Items require established work; +Open Questions require a concrete unresolved need. Unsupported candidates must +not be moved into another category. + +Action existence and responsibility attribution are separate checks. A valid +Action Item may have no known owner, while a named owner requires explicit +assignment, volunteering or acceptance. These are semantic LLM classifications; +deterministic validation must not guess intent from keywords. + ## Semantic Consolidator failure handling Semantic Consolidator V0 preserves every raw model response before parsing. @@ -396,6 +410,28 @@ an instruction not to emit an identical group more than once. Both attempts and the detected repetition metadata are preserved. If the retry also fails, the stage fails normally; it does not make another LLM call. +## Working Protocol V2 renderer contract + +The renderer deterministically projects consolidated items to the semantic +fields required for presentation and omits bulky provenance fields from the +LLM request. Every renderable item remains represented; structurally empty +items are recorded separately rather than turned into invented prose. The +compact renderer input is preserved as an artifact. + +The exact Markdown structure is generated from the same heading constants used +by the validator and appended after the renderer input so it remains visible +within the evaluated context. Decisions, action items and open questions carry +input-derived hidden coverage markers. Strict validation requires every such +renderable priority item exactly once in its matching section and rejects +missing, duplicate, wrong-section or invented markers. Facts and technical +details remain condensable as background. + +Renderer output budgeting is adaptive to required priority content and prompt +size, while an explicit `num_predict` override remains authoritative. Raw model +output is always preserved, and `working_protocol.md` is written only after +strict structure and coverage validation passes. The renderer does not retry +automatically. + --- # Next Milestone diff --git a/docs/experiments.md b/docs/experiments.md index 84b59ee..94b9db0 100644 --- a/docs/experiments.md +++ b/docs/experiments.md @@ -724,7 +724,7 @@ Evidence: ## EXP-0015 - Difficult synthetic meeting -Status: Running +Status: Accepted Date or period: 2026-07-30 @@ -1231,3 +1231,122 @@ Evidence: - `docs/output-views.md` - `prompts/working_protocol.md` - `tests/gold/responsibility_attribution_negative/` + +## EXP-0024 - Working Protocol V2 contract visibility + +Status: Running + +Date or period: 2026-08-09 + +Target: + +BUG-011 renderer-only regression using an already validated Semantic +Consolidator artifact. + +Hypothesis: + +The renderer failures are caused by provenance-heavy 199-350 KB JSON inputs +placing the leading prompt contract outside the model's effective evaluated +context. The preserved failures all report `prompt_eval_count=16386`, while +their outputs either echo trailing JSON or produce an unconstrained generic +category summary instead of the requested Working Protocol. + +Iteration 1 change: + +- Project every consolidated item to rendering-relevant semantic fields while + retaining every item and its category/text/responsibility/deadline/status + information. +- Generate the exact structural contract from renderer validator constants and + append it after the compact INPUT JSON. +- Replace the independently handwritten prompt skeleton with a reference to + that authoritative appended contract. +- Enforce the existing prompt rule that emitted sections must not be empty. + +This is one renderer-contract prompt iteration. It does not change extraction, +canonicalization, semantic consolidation or responsibility semantics. + +Validation before LLM run: + +- 15 focused renderer tests pass. +- Tests cover contract generation, compact input projection, valid and invalid + headings, missing topic sections, empty sections, wrapper cleanup, malformed + Markdown and final-file write gating. + +Decision: + +Iteration 1 passed structural validation and wrote `working_protocol.md`, but +the quality sanity check found that the model omitted most of the ten supplied +decisions, two open questions and several action items. The structurally valid +result therefore was not accepted as BUG-011 verification. + +Iteration 2 change: + +- Add input-derived hidden coverage markers for every decision, action item and + open question. +- Require every priority item exactly once in its matching section. +- Validate missing, duplicate, unknown and wrong-section markers + deterministically. +- Keep facts and technical details condensable as background. + +This is the second single prompt iteration. It responds to the concrete +omission failure observed in Iteration 1 without changing upstream semantics or +inventing renderer content. + +Iteration 2 pre-run validation: + +- 18 focused renderer tests pass, including exact required-item coverage and + wrong-section rejection. + +Decision: + +Iteration 2 initially exhausted the fixed 4,096-token renderer output budget +after emitting all decisions and most action items. Adaptive renderer budgeting +resolved to 8,192 tokens for this input. The final run stopped normally after +3,709 evaluated output tokens. + +The final renderer-only regression passed strict validation and wrote +`working_protocol.md`. Exact coverage was 10/10 decisions, 36/36 renderable +action items and 23/23 open questions, each once in its matching section. One +structurally empty action item whose task, responsible, deadline and evidence +were all null was recorded and excluded rather than fabricated. Optional +background markers were accepted only when they referred to real projected +input items. + +Accept the compact renderer input, validator-derived trailing contract, +priority-item coverage markers, empty-section validation and adaptive renderer +output sizing as the BUG-011 baseline. This establishes structural reliability +and priority-item coverage, not complete protocol prose quality. + +Evidence: + +- `prompts/working_protocol.md` +- `src/meeting_lab/protocol/render_working_protocol.py` +- `tests/test_render_working_protocol.py` +- `samples/benchmarks/progeo_qwen35_9b_20260805_133942/working_protocol/` +- `samples/benchmarks/progeo_qwen35_35b_a3b_20260805_133942/working_protocol/` + +## EXP-0025 — BUG-015 classification precision + +Date: 2026-08-09 + +Target: Progeo-derived Decision, Action Item and Open Question precision cases. + +Model/configuration: `qwen3.5:9B`, temperature 0, `num_ctx=32768`. + +Tests were created before prompt changes. A Decision-only evidence threshold +kept the explicit Dr. Schlummer rejection and omitted an option and preference. +Adding Action and Open Question definitions improved several negatives but was +not stable: the model alternately promoted an unaccepted Textor suggestion or +moved rejected candidates into Open Questions. Moving the standalone category +prompts after the transcript made the partial Decision schema dominate and +misclassified true Action Items as Decisions in two consecutive runs. + +The final iteration replaced the competing standalone category prompts with a +single unified classification contract after the transcript. It preserved the +assigned Nina task and ownerless established CET work, and prevented +cross-category leakage in the focused case, but still emitted the unaccepted +Textor suggestion as an Action Item. Further prompt iterations were stopped in +accordance with the Gold Standard methodology. + +Result: partially improved, not accepted as a complete BUG-015 fix. BUG-015 +remains Open; no phrase-specific deterministic filter was introduced. diff --git a/docs/regression-bugs.md b/docs/regression-bugs.md index 13c1af2..6aa9a0f 100644 --- a/docs/regression-bugs.md +++ b/docs/regression-bugs.md @@ -840,6 +840,51 @@ Verified. The renderer contract is now enforced deterministically. This does not improve semantic quality of the generated prose; it prevents invalid renderer output from being accepted as a final Working Protocol. +Production-blocker follow-up, 2026-08-09: + +Later preserved renderer failures showed that enforcement alone did not make a +valid protocol reliably obtainable. Provenance-heavy consolidated JSON inputs +were 199-350 KB, while failed Ollama runs consistently evaluated 16,386 prompt +tokens. The leading handwritten renderer contract was therefore effectively +lost or underweighted: models echoed trailing JSON or produced generic category +summaries. Prompt and validator also duplicated the structure independently, +and the validator did not enforce the prompt's no-empty-section rule. + +The renderer now: + +- preserves every renderable item in a compact semantic projection while + removing source-reference and original-value bulk +- records and excludes structurally empty items instead of inventing content +- appends an authoritative contract generated from validator heading constants +- rejects empty emitted sections +- requires hidden, input-derived coverage markers for every decision, action + item and open question exactly once in the matching section +- accepts optional background markers only for real projected input items +- uses adaptive output sizing instead of the truncating fixed 4,096-token cap +- keeps explicit output-budget overrides authoritative and makes no automatic + renderer retry + +Renderer-only verification used the validated consolidated input at +`/tmp/meeting-lab-bug014-regression/consolidated_extractions.json`. The final +run used `qwen3.5:9B`, temperature 0, `num_ctx=32768`, adaptive +`num_predict=8192`, and stopped normally with `eval_count=3709` and +`done_reason=stop` after 53.069 seconds. Strict validation reported no +violations and wrote: + +- `/tmp/meeting-lab-bug011-renderer-final/working_protocol.md` + +Coverage was complete for all renderable priority items: 10 decisions, 36 +action items and 23 open questions, with no missing or duplicate markers. One +upstream action item containing only null task/responsibility/deadline/evidence +was recorded as structurally empty and not rendered. The protocol retained +substantial technical background and did not add a new named responsibility; +the only structured responsible person in the renderer input remained Marleen. + +Status remains Verified for Working Protocol V2 structural validity and +priority-item coverage. This does not claim full semantic or editorial protocol +quality, topic quality, or resolution of the other renderer-related regression +bugs. + ## BUG-012 ID: BUG-012 @@ -1124,3 +1169,36 @@ source-ID occurrences and 113 unique canonical source IDs. There were no missing IDs, unknown IDs or duplicate occurrences. BUG-014 is independent of BUG-013: BUG-013 detects invalid JSON caused by runaway repeated groups, whereas BUG-014 repairs unknown source IDs in parseable model grouping JSON. + +## BUG-015 + +ID: BUG-015 + +Title: Extraction classification precision for Decisions, Action Items, and Open Questions + +Status: Open + +First observed: Progeo production benchmark and BUG-011 renderer-only output + +The Progeo extraction classified proposals/options as Decisions, suggestions +and hypothetical work as Action Items, and uncertainty or discussion fragments +as Open Questions. The production prompt combined a weak shared German rule, +a standalone Decision prompt, a responsibility-only Action prompt, and no +Open-Question definition in one multi-category extraction request. + +Regression fixtures now record explicit positive and negative evidence for all +three categories. Extraction uses one unified classification contract after the +transcript, with precision-first thresholds, category boundaries, evidence +requirements, and the responsibility-attribution invariant. In a seven-chunk +Progeo measurement, bare uncertainty in chunk 08 stopped becoming an Open +Question and the preference about real plant versus Technikum in chunk 16 +stopped becoming a Decision. Supported examples including the explicit Dr. +Schlummer rejection and established CET work remained detectable. + +BUG-015 is not Verified. With `qwen3.5:9B`, temperature 0, the focused Action +fixture still classified the unaccepted suggestion to contact Dirk Textor as +an Action Item. Earlier prompt variants also showed category leakage by turning +rejected candidates into Open Questions or Decisions. Prompt iterations were +stopped under the Gold Standard methodology rather than adding phrase-specific +filters. A general semantic classification architecture improvement remains +necessary before this bug can be marked Verified. diff --git a/prompts/classification.md b/prompts/classification.md new file mode 100644 index 0000000..867d1ff --- /dev/null +++ b/prompts/classification.md @@ -0,0 +1,55 @@ +Classification contract for Decisions, Action Items, and Open Questions: + +Classify each supported proposition into the one best category. Do not move an +unsupported candidate into another category merely to retain it. Prefer +omission over a false Decision or Action Item. + +Decision: + +A Decision requires evidence that the meeting settled, selected, approved, +rejected, or committed to an outcome. Explicit agreement, selection, +approval/rejection, commitment, or a clear statement that a decision was made +is sufficient. A proposal, suggestion, preference, possibility, option, +recommendation, consideration, or unresolved plan is not sufficient without +explicit acceptance or commitment. A personal commitment to perform work is an +Action Item, not a Decision, unless the meeting also establishes a distinct +group-level outcome. + +Action Item: + +An Action Item is a concrete future action that the meeting establishes as work +to be done. Explicit assignment, explicit acceptance, a personal commitment, +or clear evidence that the work is already established and underway is +sufficient. A suggestion, hypothetical action, possible next step, +recommendation, brainstorming option, or exploratory statement is not +sufficient. Do not create an Action Item merely because a statement contains an +imperative-like verb. + +First decide whether the action itself exists. Only then attribute +responsibility. An Action Item may exist without a named responsible person. A +person may be named only when explicitly assigned, volunteering, or explicitly +accepting/confirming the task. Mention, proximity, expertise, role, a request, +or language that someone should/could/might act does not establish +responsibility. Do not preserve an unsupported action merely by setting +responsible to null. + +Open Question: + +An Open Question requires a concrete unresolved information, decision, or +clarification need that remains open after the available context. An explicit +question left unanswered or an explicit statement that clarification remains +needed is sufficient. Uncertainty, doubt, ignorance, difficulty, an +observation, an incomplete fragment, or a rhetorical question alone is not +sufficient. A question answered in the available context is not open. Do not +invent question wording or convert a rejected Decision/Action candidate into an +Open Question without independent evidence of an unresolved need. + +Evidence discipline: + +- Every emitted item must include the shortest transcript evidence that proves + the category threshold. +- Evidence for an Action Item must prove both the concrete work and that it was + established, not merely possible. +- Evidence for an Open Question must prove both the specific need and that it + remains unresolved. +- Use an empty list when a category has no supported item. diff --git a/prompts/common.md b/prompts/common.md index 5075b8d..2cac32e 100644 --- a/prompts/common.md +++ b/prompts/common.md @@ -11,11 +11,18 @@ Regeln: 4. Wenn etwas widersprüchlich oder unklar ist, kennzeichne es als unklar. 5. Übernimm wichtige technische Aussagen möglichst nah am Wortlaut. 6. Nenne bei Fakten nach Möglichkeit den Sprecher. -7. Ein Beschluss ist nur dann ein Beschluss, wenn im Text eine Einigung, - Freigabe oder verbindliche Festlegung erkennbar ist. -8. Eine Aufgabe ist nur dann eine Aufgabe, wenn eine Handlung und möglichst - eine verantwortliche Person oder Organisation erkennbar sind. -9. Gib ausschließlich gültiges JSON aus. Kein Markdown, keine Erläuterungen. +7. Ein Beschluss ist nur dann ein Beschluss, wenn das Meeting etwas erkennbar + geeinigt, ausgewählt, freigegeben, abgelehnt oder verbindlich festgelegt hat. +8. Eine Aufgabe ist nur dann eine Aufgabe, wenn das Meeting eine konkrete + zukünftige Handlung als zu erledigende Arbeit festgelegt hat. Vorschläge, + Möglichkeiten und hypothetische nächste Schritte genügen nicht. +9. Prüfe erst, ob die Aufgabe selbst belegt ist. Ordne danach nur bei expliziter + Zuweisung oder Annahme eine verantwortliche Person zu. Eine belegte Aufgabe + darf auch keine bekannte verantwortliche Person haben. +10. Eine offene Frage erfordert einen konkreten, weiterhin offenen Informations-, + Entscheidungs- oder Klärungsbedarf. Bloße Unsicherheit genügt nicht; eine im + verfügbaren Kontext beantwortete Frage bleibt nicht offen. +11. Gib ausschließlich gültiges JSON aus. Kein Markdown, keine Erläuterungen. Hinweise zur Ausgabe: - Alle obersten Schlüssel müssen vorhanden sein. diff --git a/prompts/questions.md b/prompts/questions.md deleted file mode 100644 index e69de29..0000000 diff --git a/src/meeting_lab/extraction/extract_chunks.py b/src/meeting_lab/extraction/extract_chunks.py index c2ddbda..2496b19 100644 --- a/src/meeting_lab/extraction/extract_chunks.py +++ b/src/meeting_lab/extraction/extract_chunks.py @@ -41,7 +41,7 @@ EXTRACTION_CATEGORIES = ( NORMALIZED_CHUNK_RE = re.compile(r"^(chunk_\d+)_normalized\.txt$") -EXTRACTION_TASK_PROMPT_NAMES = ("decisions.md", "todos.md") +EXTRACTION_TASK_PROMPT_NAMES = ("classification.md",) OUTPUT_SCHEMA = { diff --git a/src/meeting_lab/llm/prompts.py b/src/meeting_lab/llm/prompts.py index 3be33a6..d0cc53c 100644 --- a/src/meeting_lab/llm/prompts.py +++ b/src/meeting_lab/llm/prompts.py @@ -38,7 +38,6 @@ def build_extraction_prompt( prompt_parts = [ load_prompt("common.md", prompts_dir), meeting_context, - *load_existing_prompts(task_prompt_names, prompts_dir), f"""Quelldatei: {source_name} @@ -49,5 +48,6 @@ TRANSKRIPT: --- BEGINN TRANSKRIPT --- {transcript} --- ENDE TRANSKRIPT ---""", + *load_existing_prompts(task_prompt_names, prompts_dir), ] return "\n\n".join(part for part in prompt_parts if part).strip() + "\n" diff --git a/tests/gold/progeo_action_precision/README.md b/tests/gold/progeo_action_precision/README.md new file mode 100644 index 0000000..97141c1 --- /dev/null +++ b/tests/gold/progeo_action_precision/README.md @@ -0,0 +1,8 @@ +# progeo_action_precision + +Minimal German regression derived from Progeo chunks 2, 10, 17 and 18. It +separates suggestions and exploratory possibilities from established work and +tests explicit acceptance of responsibility. + +Expected semantics: only the accepted review task and the already-established +article work are Action Items. diff --git a/tests/gold/progeo_action_precision/expected.json b/tests/gold/progeo_action_precision/expected.json new file mode 100644 index 0000000..fb81f34 --- /dev/null +++ b/tests/gold/progeo_action_precision/expected.json @@ -0,0 +1,21 @@ +{ + "facts": [], + "decisions": [], + "todos": [ + { + "task": "Messdaten bis Freitag prüfen.", + "responsible": "Nina", + "deadline": "Freitag", + "evidence": "Nina: Ja, ich übernehme die Prüfung bis Freitag." + }, + { + "task": "CET-Artikel erstellen und anschließend im Konsortium verteilen.", + "responsible": null, + "deadline": null, + "evidence": "Den CET-Artikel erstellen wir bereits und schicken ihn anschließend im Konsortium herum." + } + ], + "questions": [], + "positions": [], + "technical": [] +} diff --git a/tests/gold/progeo_action_precision/transcript.txt b/tests/gold/progeo_action_precision/transcript.txt new file mode 100644 index 0000000..f6fdce1 --- /dev/null +++ b/tests/gold/progeo_action_precision/transcript.txt @@ -0,0 +1,11 @@ +Martin: Die Geometrie kann man vielleicht noch optimieren. Dann würde man mal gucken, was herauskommt. + +Antonius: Marleen müsste vielleicht mal äußern, was Hüsker sich unter dem Versuch vorstellt. + +Tim: Wir könnten Dirk Textor vielleicht noch einmal kontaktieren. + +Antonius: Nina, übernimmst du die Prüfung der Messdaten bis Freitag? + +Nina: Ja, ich übernehme die Prüfung bis Freitag. + +Tim: Den CET-Artikel erstellen wir bereits und schicken ihn anschließend im Konsortium herum. diff --git a/tests/gold/progeo_decision_precision/README.md b/tests/gold/progeo_decision_precision/README.md new file mode 100644 index 0000000..4eb5734 --- /dev/null +++ b/tests/gold/progeo_decision_precision/README.md @@ -0,0 +1,6 @@ +# progeo_decision_precision + +Minimal German regression derived from Progeo chunks 12 and 16. It contrasts +an option and a personal preference with an explicit group rejection. + +Expected semantics: only the explicit rejection is a Decision. diff --git a/tests/gold/progeo_decision_precision/expected.json b/tests/gold/progeo_decision_precision/expected.json new file mode 100644 index 0000000..7516d51 --- /dev/null +++ b/tests/gold/progeo_decision_precision/expected.json @@ -0,0 +1,13 @@ +{ + "facts": [], + "decisions": [ + { + "decision": "Die Gruppe lehnt die Zusammenarbeit mit Dr. Schlummer ab.", + "evidence": "Dann haben wir gesagt: Nein, die Zusammenarbeit mit Dr. Schlummer machen wir nicht. Antonius: Ja, das ist entschieden." + } + ], + "todos": [], + "questions": [], + "positions": [], + "technical": [] +} diff --git a/tests/gold/progeo_decision_precision/transcript.txt b/tests/gold/progeo_decision_precision/transcript.txt new file mode 100644 index 0000000..4e45602 --- /dev/null +++ b/tests/gold/progeo_decision_precision/transcript.txt @@ -0,0 +1,9 @@ +Tim: Die Option steht im Raum, das Material chemisch recyceln zu lassen. + +Martin: Ich würde nicht in eine reale Anlage gehen. Wenn überhaupt, können wir über ein Technikum reden. + +Antonius: Das Angebot von Dr. Schlummer kostet 30.000 Euro. + +Tim: Dann haben wir gesagt: Nein, die Zusammenarbeit mit Dr. Schlummer machen wir nicht. + +Antonius: Ja, das ist entschieden. diff --git a/tests/gold/progeo_question_precision/README.md b/tests/gold/progeo_question_precision/README.md new file mode 100644 index 0000000..3de3171 --- /dev/null +++ b/tests/gold/progeo_question_precision/README.md @@ -0,0 +1,8 @@ +# progeo_question_precision + +Minimal German regression derived from Progeo chunks 8 and 18. It contrasts +uncertainty, an observation and an immediately answered question with an +explicitly unresolved information need. + +Expected semantics: only the explicitly unresolved publication question is an +Open Question. diff --git a/tests/gold/progeo_question_precision/expected.json b/tests/gold/progeo_question_precision/expected.json new file mode 100644 index 0000000..4f0071b --- /dev/null +++ b/tests/gold/progeo_question_precision/expected.json @@ -0,0 +1,13 @@ +{ + "facts": [], + "decisions": [], + "todos": [], + "questions": [ + { + "question": "Welche Daten aus dem Energieaudit dürfen veröffentlicht werden?", + "evidence": "Welche Daten aus dem Energieaudit dürfen wir veröffentlichen? Martin: Das ist weiterhin ungeklärt." + } + ], + "positions": [], + "technical": [] +} diff --git a/tests/gold/progeo_question_precision/transcript.txt b/tests/gold/progeo_question_precision/transcript.txt new file mode 100644 index 0000000..27fbd6d --- /dev/null +++ b/tests/gold/progeo_question_precision/transcript.txt @@ -0,0 +1,11 @@ +Martin: Ob sich das Waschen lohnt, weiß ich nicht. + +Tim: Das lässt sich teilweise optimieren und teilweise nicht. + +Antonius: Gibt es schon ein Programm für die Veranstaltung? + +Tim: Ja, das Programm steht auf der Dechema-Seite und ist vollständig. + +Antonius: Welche Daten aus dem Energieaudit dürfen wir veröffentlichen? + +Martin: Das ist weiterhin ungeklärt. Wir müssen die Freigabe noch klären. diff --git a/tests/test_classification_invariants.py b/tests/test_classification_invariants.py new file mode 100644 index 0000000..cd33b68 --- /dev/null +++ b/tests/test_classification_invariants.py @@ -0,0 +1,36 @@ +import unittest + +from src.meeting_lab.extraction.extract_chunks import build_prompt + + +class ClassificationInvariantPromptTests(unittest.TestCase): + def setUp(self) -> None: + self.prompt = build_prompt("regression.txt", "Meeting transcript.") + + def test_decision_requires_settled_outcome_and_prefers_precision(self) -> None: + self.assertIn("settled, selected, approved", self.prompt) + self.assertIn("possibility, option", self.prompt) + self.assertIn("Prefer\nomission over a false Decision or Action Item", self.prompt) + + def test_action_requires_established_future_work(self) -> None: + self.assertIn("concrete future action that the meeting establishes as", self.prompt) + self.assertIn("suggestion, hypothetical action, possible next step", self.prompt.lower()) + self.assertIn("Do not create an Action Item merely because", self.prompt) + self.assertIn("statement contains an", self.prompt) + self.assertIn("imperative-like verb", self.prompt) + + def test_action_existence_is_separate_from_responsibility(self) -> None: + self.assertIn("First decide whether the action itself exists", self.prompt) + self.assertIn("An Action Item may exist without a named responsible", self.prompt) + self.assertIn("responsible person", self.prompt) + self.assertIn("explicitly assigned, volunteering", self.prompt) + self.assertIn("accepting/confirming the task", self.prompt) + + def test_open_question_requires_unresolved_need(self) -> None: + self.assertIn("concrete unresolved information, decision, or", self.prompt) + self.assertIn("Uncertainty, doubt, ignorance, difficulty", self.prompt) + self.assertIn("answered in the available context is not open", self.prompt) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_extraction_protocol.py b/tests/test_extraction_protocol.py index 605982e..9518432 100644 --- a/tests/test_extraction_protocol.py +++ b/tests/test_extraction_protocol.py @@ -101,17 +101,20 @@ Final answer: prompt = build_prompt("transcript.txt", "Anna: Agreed.") self.assertIn("Du extrahierst Informationen aus Meeting-Transkripten.", prompt) - self.assertIn("You extract decisions from meeting transcript text.", prompt) - self.assertIn("Extract each decision as one atomic commitment.", prompt) + self.assertIn("Classification contract for Decisions", prompt) + self.assertIn("A Decision requires evidence", prompt) self.assertIn("Anna: Agreed.", prompt) def test_build_prompt_includes_todo_prompt_file(self) -> None: prompt = build_prompt("transcript.txt", "Nina: I will update it.") - self.assertEqual(EXTRACTION_TASK_PROMPT_NAMES, ("decisions.md", "todos.md")) - self.assertIn("Action-item responsibility rule:", prompt) + self.assertEqual( + EXTRACTION_TASK_PROMPT_NAMES, + ("classification.md",), + ) + self.assertIn("Action Item:", prompt) self.assertIn( - "A named responsible person may be extracted only when the transcript contains", + "person may be named only when explicitly assigned", prompt, ) @@ -131,8 +134,9 @@ Final answer: ) for prompt in (no_context_prompt, context_prompt): - self.assertIn("You extract decisions from meeting transcript text.", prompt) - self.assertIn("Action-item responsibility rule:", prompt) + self.assertIn("Classification contract for Decisions", prompt) + self.assertIn("Action Item:", prompt) + self.assertIn("Open Question:", prompt) def test_prompt_assembly_is_deterministic(self) -> None: first = build_prompt("transcript.txt", "Nina: I will update it.") @@ -140,6 +144,20 @@ Final answer: self.assertEqual(first, second) + def test_classification_contracts_follow_transcript(self) -> None: + prompt = build_prompt("transcript.txt", "SOURCE_SENTINEL") + + transcript_end = prompt.index("--- ENDE TRANSKRIPT ---") + self.assertGreater(prompt.index("Classification contract", transcript_end), transcript_end) + self.assertGreater( + prompt.index("Action Item:", transcript_end), + transcript_end, + ) + self.assertGreater( + prompt.index("Open Question:", transcript_end), + transcript_end, + ) + def test_build_protocol_groups_extraction_items(self) -> None: with tempfile.TemporaryDirectory() as directory: input_dir = Path(directory)