From f7ad9ba51f1cec3b09b4075d1fd77b70e6d1fa91 Mon Sep 17 00:00:00 2001 From: Martin Tazl Date: Thu, 30 Jul 2026 12:13:10 +0200 Subject: [PATCH] Establish prompt engineering baseline with Gold Standard tests - introduce Gold Standard evaluation corpus - document decision taxonomy - define prompt-engineering methodology - add regression workflow - establish Prompt Version 2 baseline - validate decision_simple, decision_deferred and decision_none --- .gitignore | 1 + prompts/common.md | 26 +++ prompts/decisions.md | 86 +++++++ scripts/run_gold_test.py | 223 +++++++++++++++++++ src/meeting_lab/extraction/extract_chunks.py | 48 +--- src/meeting_lab/llm/prompts.py | 51 +++++ tests/gold/DECISION_DEFINITION.md | 32 +++ tests/gold/PROMPT_ENGINEERING_METHODOLOGY.md | 34 +++ tests/gold/decision_deferred/README.md | 21 ++ tests/gold/decision_deferred/expected.json | 31 +++ tests/gold/decision_deferred/transcript.txt | 19 ++ tests/gold/decision_none/README.md | 21 ++ tests/gold/decision_none/expected.json | 24 ++ tests/gold/decision_none/transcript.txt | 25 +++ tests/gold/decision_simple/README.md | 11 + tests/gold/decision_simple/expected.json | 13 ++ tests/gold/decision_simple/transcript.txt | 19 ++ tests/gold/evil_meeting/README.md | 14 ++ tests/gold/evil_meeting/expected.json | 73 ++++++ tests/gold/evil_meeting/transcript.txt | 61 +++++ tests/gold/facts_simple/README.md | 11 + tests/gold/facts_simple/expected.json | 32 +++ tests/gold/facts_simple/transcript.txt | 19 ++ tests/gold/facts_vs_positions/README.md | 11 + tests/gold/facts_vs_positions/expected.json | 42 ++++ tests/gold/facts_vs_positions/transcript.txt | 19 ++ tests/gold/mixed_small/README.md | 11 + tests/gold/mixed_small/expected.json | 50 +++++ tests/gold/mixed_small/transcript.txt | 23 ++ tests/gold/question_simple/README.md | 11 + tests/gold/question_simple/expected.json | 27 +++ tests/gold/question_simple/transcript.txt | 19 ++ tests/gold/technical_simple/README.md | 11 + tests/gold/technical_simple/expected.json | 33 +++ tests/gold/technical_simple/transcript.txt | 19 ++ tests/gold/todo_negative/README.md | 11 + tests/gold/todo_negative/expected.json | 22 ++ tests/gold/todo_negative/transcript.txt | 19 ++ tests/gold/todo_simple/README.md | 11 + tests/gold/todo_simple/expected.json | 20 ++ tests/gold/todo_simple/transcript.txt | 19 ++ tests/test_extraction_protocol.py | 9 + tests/test_gold_runner.py | 51 +++++ 43 files changed, 1288 insertions(+), 45 deletions(-) create mode 100644 prompts/common.md create mode 100644 scripts/run_gold_test.py create mode 100644 tests/gold/DECISION_DEFINITION.md create mode 100644 tests/gold/PROMPT_ENGINEERING_METHODOLOGY.md create mode 100644 tests/gold/decision_deferred/README.md create mode 100644 tests/gold/decision_deferred/expected.json create mode 100644 tests/gold/decision_deferred/transcript.txt create mode 100644 tests/gold/decision_none/README.md create mode 100644 tests/gold/decision_none/expected.json create mode 100644 tests/gold/decision_none/transcript.txt create mode 100644 tests/gold/decision_simple/README.md create mode 100644 tests/gold/decision_simple/expected.json create mode 100644 tests/gold/decision_simple/transcript.txt create mode 100644 tests/gold/evil_meeting/README.md create mode 100644 tests/gold/evil_meeting/expected.json create mode 100644 tests/gold/evil_meeting/transcript.txt create mode 100644 tests/gold/facts_simple/README.md create mode 100644 tests/gold/facts_simple/expected.json create mode 100644 tests/gold/facts_simple/transcript.txt create mode 100644 tests/gold/facts_vs_positions/README.md create mode 100644 tests/gold/facts_vs_positions/expected.json create mode 100644 tests/gold/facts_vs_positions/transcript.txt create mode 100644 tests/gold/mixed_small/README.md create mode 100644 tests/gold/mixed_small/expected.json create mode 100644 tests/gold/mixed_small/transcript.txt create mode 100644 tests/gold/question_simple/README.md create mode 100644 tests/gold/question_simple/expected.json create mode 100644 tests/gold/question_simple/transcript.txt create mode 100644 tests/gold/technical_simple/README.md create mode 100644 tests/gold/technical_simple/expected.json create mode 100644 tests/gold/technical_simple/transcript.txt create mode 100644 tests/gold/todo_negative/README.md create mode 100644 tests/gold/todo_negative/expected.json create mode 100644 tests/gold/todo_negative/transcript.txt create mode 100644 tests/gold/todo_simple/README.md create mode 100644 tests/gold/todo_simple/expected.json create mode 100644 tests/gold/todo_simple/transcript.txt create mode 100644 tests/test_gold_runner.py diff --git a/.gitignore b/.gitignore index a516c1d..f2b3526 100644 --- a/.gitignore +++ b/.gitignore @@ -38,6 +38,7 @@ samples/whisper/ **/chunk_*_extraction.json **/meeting_protocol.md **/*.raw.txt +tests/gold/**/actual.json # Lokale Meetings (niemals versionieren) meeting_data/ diff --git a/prompts/common.md b/prompts/common.md new file mode 100644 index 0000000..5075b8d --- /dev/null +++ b/prompts/common.md @@ -0,0 +1,26 @@ +Du extrahierst Informationen aus Meeting-Transkripten. + +Arbeite ausschließlich mit dem vorgelegten Transkript. +Verwende kein eigenes Fachwissen, keine Vermutungen und keine üblichen +Funktionsweisen technischer Systeme. + +Regeln: +1. Erfinde nichts. +2. Interpretiere technische Aussagen nicht über den Wortlaut hinaus. +3. Korrigiere keine Aussagen anhand vermeintlichen Weltwissens. +4. Wenn etwas widersprüchlich oder unklar ist, kennzeichne es als unklar. +5. Übernimm wichtige technische Aussagen möglichst nah am Wortlaut. +6. Nenne bei Fakten nach Möglichkeit den Sprecher. +7. Ein Beschluss ist nur dann ein Beschluss, wenn im Text eine Einigung, + Freigabe oder verbindliche Festlegung erkennbar ist. +8. Eine Aufgabe ist nur dann eine Aufgabe, wenn eine Handlung und möglichst + eine verantwortliche Person oder Organisation erkennbar sind. +9. Gib ausschließlich gültiges JSON aus. Kein Markdown, keine Erläuterungen. + +Hinweise zur Ausgabe: +- Alle obersten Schlüssel müssen vorhanden sein. +- Verwende leere Listen, wenn keine Einträge vorhanden sind. +- Verwende null, wenn Verantwortliche, Sprecher oder Termine nicht erkennbar sind. +- "evidence" muss sich eng am Transkript orientieren. +- Ersetze technische Aussagen niemals durch eine vermeintlich korrektere Erklärung. +- Confidence-Werte sind ausdrücklich nicht erwünscht. diff --git a/prompts/decisions.md b/prompts/decisions.md index e69de29..ca73344 100644 --- a/prompts/decisions.md +++ b/prompts/decisions.md @@ -0,0 +1,86 @@ +You extract decisions from meeting transcript text. + +Return only valid JSON. + +Schema: + +{ + "decisions": [ + { + "decision": "Concise decision text", + "evidence": "Short quote from the transcript" + } + ] +} + +Definition: + +A decision exists only when the participants explicitly agreed, approved, +confirmed, adopted, assigned, or otherwise made something binding during the +meeting. + +Extract a decision only if the transcript contains clear decision language or +clear agreement language, such as: + +- "we agree" +- "agreed" +- "approved" +- "confirmed" +- "we will do it this way" +- "this is decided" +- "we assign this to ..." +- "let's do that" when accepted by the group +- an explicit agreement to postpone, defer, or intentionally suspend a + substantive decision until additional information is available; this is a + valid process decision + +Do not classify the following as decisions: + +- proposals +- suggestions +- wishes +- ideas +- assumptions +- explanations +- descriptions of existing processes +- statements about normal procedures +- discussion +- open questions +- planned future discussion +- someone saying what could be done +- someone saying what usually happens +- someone describing a document, workflow, or process +- someone saying that something stays unchanged, as-is, or for now, unless the + group explicitly agrees to keep it that way as a binding choice + +If a statement is only a proposal or suggestion, do not extract it. + +If participants discuss something but do not explicitly agree to it, do not +extract it. + +If the transcript describes an existing process, rule, template, document, or +workflow, do not extract it unless the participants explicitly adopt or change it +in this meeting. + +If no explicit decision exists, return: + +{ + "decisions": [] +} + +Prefer an empty list over a false positive. + +For each decision: + +- Write one concise decision text. +- Extract each decision as one atomic commitment. +- Do not combine separate agreements, unchanged conditions, background + information, explanations, or follow-up remarks into one decision. +- If two distinct matters were agreed, return two separate decisions. +- Include one short evidence quote from the transcript. +- Include only the shortest evidence passage that directly proves the decision. +- Do not invent responsible persons. +- Do not invent deadlines. +- Do not invent priorities. +- Do not add confidence values. +- Do not explain your reasoning. diff --git a/scripts/run_gold_test.py b/scripts/run_gold_test.py new file mode 100644 index 0000000..1e07e81 --- /dev/null +++ b/scripts/run_gold_test.py @@ -0,0 +1,223 @@ +#!/usr/bin/env python3 +"""Run one gold-corpus scenario through the existing extraction flow.""" + +from __future__ import annotations + +import argparse +import json +import sys +import time +from pathlib import Path +from typing import Any + +import requests + +REPO_ROOT = Path(__file__).resolve().parents[1] +if str(REPO_ROOT) not in sys.path: + sys.path.insert(0, str(REPO_ROOT)) + +from src.meeting_lab.extraction.extract_chunks import ( + DEFAULT_ENDPOINT, + EXTRACTION_CATEGORIES, + build_prompt, + call_ollama, + normalize_current_schema, + parse_json_response, +) + + +REQUIRED_KEYS = set(EXTRACTION_CATEGORIES) + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser( + description="Run one gold-standard transcript through extraction." + ) + parser.add_argument("scenario", type=Path, help="Gold scenario directory") + parser.add_argument("--model", required=True, help="Ollama model name") + parser.add_argument( + "--endpoint", + default=DEFAULT_ENDPOINT, + help=f"Ollama generate endpoint (default: {DEFAULT_ENDPOINT})", + ) + parser.add_argument( + "--timeout", + type=int, + default=1800, + help="HTTP timeout in seconds (default: 1800)", + ) + parser.add_argument( + "--temperature", + type=float, + default=0.0, + help="Sampling temperature (default: 0.0)", + ) + parser.add_argument( + "--num-predict", + type=int, + default=8192, + help="Maximum generated tokens (default: 8192)", + ) + parser.add_argument( + "--num-ctx", + type=int, + default=32768, + help="Context window tokens (default: 32768)", + ) + return parser.parse_args() + + +def scenario_paths(scenario_dir: Path) -> tuple[Path, Path, Path]: + if not scenario_dir.is_dir(): + raise FileNotFoundError(f"Scenario directory not found: {scenario_dir}") + + transcript_path = scenario_dir / "transcript.txt" + expected_path = scenario_dir / "expected.json" + actual_path = scenario_dir / "actual.json" + + if not transcript_path.is_file(): + raise FileNotFoundError(f"Missing transcript.txt: {transcript_path}") + if not expected_path.is_file(): + raise FileNotFoundError(f"Missing expected.json: {expected_path}") + + return transcript_path, expected_path, actual_path + + +def read_json_object(path: Path) -> dict[str, Any]: + try: + data = json.loads(path.read_text(encoding="utf-8")) + except json.JSONDecodeError as exc: + raise ValueError(f"Invalid JSON in {path}: {exc}") from exc + + if not isinstance(data, dict): + raise ValueError(f"JSON file must contain an object: {path}") + + return data + + +def validate_required_keys(data: dict[str, Any], path: Path) -> None: + missing = sorted(REQUIRED_KEYS - set(data)) + if missing: + raise ValueError(f"Missing required keys in {path}: {', '.join(missing)}") + + +def format_items(items: Any) -> list[str]: + if not isinstance(items, list): + return [f""] + if not items: + return [""] + return [str(item) for item in items] + + +def print_decision_comparison( + scenario_dir: Path, + model: str, + expected: dict[str, Any], + actual: dict[str, Any], + runtime: float, +) -> None: + expected_decisions = expected.get("decisions", []) + actual_decisions = actual.get("decisions", []) + + print(f"Scenario: {scenario_dir}") + print(f"Model: {model}") + print(f"Expected decision count: {len(expected_decisions)}") + print(f"Actual decision count: {len(actual_decisions)}") + print("Expected decisions:") + for item in format_items(expected_decisions): + print(f"- {item}") + print("Actual decisions:") + for item in format_items(actual_decisions): + print(f"- {item}") + print(f"Runtime: {runtime:.2f}s") + + +def run_gold_test( + scenario_dir: Path, + model: str, + endpoint: str, + timeout: int, + temperature: float, + num_predict: int | None, + num_ctx: int | None, +) -> Path: + transcript_path, expected_path, actual_path = scenario_paths(scenario_dir) + expected = read_json_object(expected_path) + validate_required_keys(expected, expected_path) + + transcript = transcript_path.read_text(encoding="utf-8-sig").strip() + if not transcript: + raise ValueError(f"The transcript is empty: {transcript_path}") + + prompt = build_prompt(transcript_path.name, transcript) + started = time.perf_counter() + raw_text, _metadata = call_ollama( + endpoint=endpoint, + model=model, + prompt=prompt, + timeout=timeout, + temperature=temperature, + num_predict=num_predict, + num_ctx=num_ctx, + ) + + try: + parsed = parse_json_response(raw_text) + except (json.JSONDecodeError, ValueError) as exc: + raw_path = actual_path.with_suffix(".raw.txt") + raw_path.write_text(raw_text + "\n", encoding="utf-8") + raise ValueError(f"Model output was not valid JSON. Raw output: {raw_path}") from exc + + actual = normalize_current_schema(parsed) + actual_path.write_text( + json.dumps(actual, ensure_ascii=False, indent=2) + "\n", + encoding="utf-8", + ) + + actual_from_disk = read_json_object(actual_path) + validate_required_keys(actual_from_disk, actual_path) + runtime = time.perf_counter() - started + print_decision_comparison( + scenario_dir=scenario_dir, + model=model, + expected=expected, + actual=actual_from_disk, + runtime=runtime, + ) + return actual_path + + +def main() -> int: + args = parse_args() + try: + actual_path = run_gold_test( + scenario_dir=args.scenario, + model=args.model, + endpoint=args.endpoint, + timeout=args.timeout, + temperature=args.temperature, + num_predict=args.num_predict, + num_ctx=args.num_ctx, + ) + except requests.ConnectionError: + print( + "Error: Ollama is not reachable. Is `ollama serve` running?", + file=sys.stderr, + ) + return 1 + except requests.Timeout: + print("Error: The Ollama request timed out.", file=sys.stderr) + return 1 + except requests.HTTPError as exc: + print(f"Error: Ollama returned an HTTP error: {exc}", file=sys.stderr) + return 1 + except (OSError, UnicodeError, ValueError) as exc: + print(f"Error: {exc}", file=sys.stderr) + return 1 + + print(f"Actual JSON: {actual_path}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/meeting_lab/extraction/extract_chunks.py b/src/meeting_lab/extraction/extract_chunks.py index f51f6f0..f5bfaa8 100644 --- a/src/meeting_lab/extraction/extract_chunks.py +++ b/src/meeting_lab/extraction/extract_chunks.py @@ -19,6 +19,8 @@ from typing import Any import requests +from src.meeting_lab.llm.prompts import build_extraction_prompt + DEFAULT_MODEL = "qwen3:8b" DEFAULT_ENDPOINT = "http://localhost:11434/api/generate" @@ -35,28 +37,6 @@ EXTRACTION_CATEGORIES = ( NORMALIZED_CHUNK_RE = re.compile(r"^(chunk_\d+)_normalized\.txt$") -SYSTEM_INSTRUCTION = """ -Du extrahierst Informationen aus Meeting-Transkripten. - -Arbeite ausschließlich mit dem vorgelegten Transkript. -Verwende kein eigenes Fachwissen, keine Vermutungen und keine üblichen -Funktionsweisen technischer Systeme. - -Regeln: -1. Erfinde nichts. -2. Interpretiere technische Aussagen nicht über den Wortlaut hinaus. -3. Korrigiere keine Aussagen anhand vermeintlichen Weltwissens. -4. Wenn etwas widersprüchlich oder unklar ist, kennzeichne es als unklar. -5. Übernimm wichtige technische Aussagen möglichst nah am Wortlaut. -6. Nenne bei Fakten nach Möglichkeit den Sprecher. -7. Ein Beschluss ist nur dann ein Beschluss, wenn im Text eine Einigung, - Freigabe oder verbindliche Festlegung erkennbar ist. -8. Eine Aufgabe ist nur dann eine Aufgabe, wenn eine Handlung und möglichst - eine verantwortliche Person oder Organisation erkennbar sind. -9. Gib ausschließlich gültiges JSON aus. Kein Markdown, keine Erläuterungen. -""".strip() - - OUTPUT_SCHEMA = { "chunk": { "source_file": "string", @@ -162,29 +142,7 @@ def parse_args() -> argparse.Namespace: def build_prompt(source_name: str, transcript: str) -> str: - schema_text = json.dumps(OUTPUT_SCHEMA, ensure_ascii=False, indent=2) - - return f"""{SYSTEM_INSTRUCTION} - -Quelldatei: -{source_name} - -Erwartete JSON-Struktur: -{schema_text} - -Hinweise zur Ausgabe: -- Alle obersten Schlüssel müssen vorhanden sein. -- Verwende leere Listen, wenn keine Einträge vorhanden sind. -- Verwende null, wenn Verantwortliche, Sprecher oder Termine nicht erkennbar sind. -- "evidence" muss sich eng am Transkript orientieren. -- Ersetze technische Aussagen niemals durch eine vermeintlich korrektere Erklärung. -- Confidence-Werte sind ausdrücklich nicht erwünscht. - -TRANSKRIPT: ---- BEGINN TRANSKRIPT --- -{transcript} ---- ENDE TRANSKRIPT --- -""" + return build_extraction_prompt(source_name, transcript, OUTPUT_SCHEMA) def call_ollama( diff --git a/src/meeting_lab/llm/prompts.py b/src/meeting_lab/llm/prompts.py index e69de29..13c95f0 100644 --- a/src/meeting_lab/llm/prompts.py +++ b/src/meeting_lab/llm/prompts.py @@ -0,0 +1,51 @@ +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any, Iterable + + +PROJECT_ROOT = Path(__file__).resolve().parents[3] +PROMPTS_DIR = PROJECT_ROOT / "prompts" + + +def load_prompt(name: str, prompts_dir: Path = PROMPTS_DIR) -> str: + path = prompts_dir / name + return path.read_text(encoding="utf-8").strip() + + +def load_existing_prompts( + names: Iterable[str], + prompts_dir: Path = PROMPTS_DIR, +) -> list[str]: + prompts: list[str] = [] + for name in names: + text = load_prompt(name, prompts_dir) + if text: + prompts.append(text) + return prompts + + +def build_extraction_prompt( + source_name: str, + transcript: str, + output_schema: dict[str, Any], + task_prompt_names: Iterable[str] = ("decisions.md",), + prompts_dir: Path = PROMPTS_DIR, +) -> str: + schema_text = json.dumps(output_schema, ensure_ascii=False, indent=2) + prompt_parts = [ + load_prompt("common.md", prompts_dir), + *load_existing_prompts(task_prompt_names, prompts_dir), + f"""Quelldatei: +{source_name} + +Erwartete JSON-Struktur: +{schema_text} + +TRANSKRIPT: +--- BEGINN TRANSKRIPT --- +{transcript} +--- ENDE TRANSKRIPT ---""", + ] + return "\n\n".join(part for part in prompt_parts if part).strip() + "\n" diff --git a/tests/gold/DECISION_DEFINITION.md b/tests/gold/DECISION_DEFINITION.md new file mode 100644 index 0000000..d1177fc --- /dev/null +++ b/tests/gold/DECISION_DEFINITION.md @@ -0,0 +1,32 @@ +# Decision Definition + +A decision is any explicit agreement that creates a binding change in action, +process, responsibility, approval status, timing, or next step. + +Included: + +- substantive decisions +- organizational decisions +- process decisions +- approvals +- rejections +- deferrals +- explicit agreement not to decide yet +- explicit agreement to gather more information before deciding + +Excluded: + +- opinions +- preferences +- proposals without agreement +- open questions +- descriptions of the current state +- explanations without commitment + +Important distinction: + +- "No decision was reached" means the meeting ended without an agreed outcome. +- "The group decided to defer the decision" means the group explicitly agreed + on a process outcome: the substantive decision is postponed. + +These are not equivalent. diff --git a/tests/gold/PROMPT_ENGINEERING_METHODOLOGY.md b/tests/gold/PROMPT_ENGINEERING_METHODOLOGY.md new file mode 100644 index 0000000..20e612b --- /dev/null +++ b/tests/gold/PROMPT_ENGINEERING_METHODOLOGY.md @@ -0,0 +1,34 @@ +# Prompt Engineering Methodology + +Prompt engineering uses the Gold Standard corpus as the reference. The corpus is +not adjusted to make a prompt pass. + +Rules: + +1. Only one prompt change is allowed per iteration. +2. Only one gold test case may be optimized at a time. +3. Every prompt modification must be validated immediately. +4. A prompt modification is acceptable only if it improves the current target + and does not degrade any previously passing gold test. +5. Never modify `expected.json` to make a prompt pass. +6. Prompt engineering edits prompt files only. Python code changes require a + separate explicit task. +7. Maintain a prompt evolution log for every iteration. +8. If a prompt cannot improve a test after several small iterations, stop and + analyze the root cause. +9. Prompt changes must be generally applicable and must not special-case one + transcript. +10. If two consecutive prompt iterations fail to improve the current target, + stop further prompt modifications and classify the root cause. +11. If a prompt produces unexpected behavior, first verify whether the targeted + gold test has an objectively unique ground truth. + +Prompt Version 2 baseline: + +- `decision_simple`: passing +- `decision_deferred`: passing +- `decision_none`: passing + +Prompt Version 2 adds explicit support for process decisions where the group +agrees to defer a substantive decision until additional information is +available. diff --git a/tests/gold/decision_deferred/README.md b/tests/gold/decision_deferred/README.md new file mode 100644 index 0000000..100d417 --- /dev/null +++ b/tests/gold/decision_deferred/README.md @@ -0,0 +1,21 @@ +# decision_deferred + +Tests that a process decision to defer a substantive decision is still extracted +as a decision. + +The transcript contains several candidate options for the weekly dashboard, but +the group does not choose any of them. Instead, Mira explicitly says not to +decide today and Jonas agrees that more input is needed. + +Ground truth: + +- No substantive dashboard schedule decision is reached. +- One process decision is reached: the substantive decision is deferred until + more information is available. + +Typical LLM mistakes: + +- Extracting Monday, Tuesday, or Friday as the chosen dashboard day. +- Treating "I like shorter" as an approval. +- Missing the deferral because the substantive decision is unresolved. +- Treating "no decision today" as equivalent to no decision at all. diff --git a/tests/gold/decision_deferred/expected.json b/tests/gold/decision_deferred/expected.json new file mode 100644 index 0000000..ae000c1 --- /dev/null +++ b/tests/gold/decision_deferred/expected.json @@ -0,0 +1,31 @@ +{ + "facts": [], + "decisions": [ + { + "decision": "The substantive decision is deferred until more information is available.", + "evidence": "Mira: Okay, let's not decide this today. Jonas: Agreed, we need more input." + } + ], + "todos": [ + { + "task": "Lea will bring Dana's feedback about the weekly dashboard next time.", + "responsible": "Lea", + "deadline": "next time", + "evidence": "Lea: I will bring Dana's feedback next time." + } + ], + "questions": [], + "positions": [ + { + "speaker": "Jonas", + "position": "Jonas prefers moving the weekly dashboard to Monday morning.", + "evidence": "Jonas: I would prefer moving it to Monday morning." + }, + { + "speaker": "Lea", + "position": "Lea thinks Monday is difficult for support.", + "evidence": "Lea: Monday is rough for support." + } + ], + "technical": [] +} diff --git a/tests/gold/decision_deferred/transcript.txt b/tests/gold/decision_deferred/transcript.txt new file mode 100644 index 0000000..c1b3cc6 --- /dev/null +++ b/tests/gold/decision_deferred/transcript.txt @@ -0,0 +1,19 @@ +Mira: We need to talk about the weekly dashboard. + +Jonas: I would prefer moving it to Monday morning. + +Lea: Monday is rough for support. We usually have backlog cleanup then. + +Mira: Tuesday might work, but I am not sure. + +Jonas: Or we keep it on Friday and just make it shorter. + +Lea: I like shorter, but I need to check with Dana first. + +Mira: Okay, let's not decide this today. + +Jonas: Agreed, we need more input. + +Lea: I will bring Dana's feedback next time. + +Mira: Thanks, that will help. diff --git a/tests/gold/decision_none/README.md b/tests/gold/decision_none/README.md new file mode 100644 index 0000000..97b7e5b --- /dev/null +++ b/tests/gold/decision_none/README.md @@ -0,0 +1,21 @@ +# decision_none + +Tests a true decision-negative meeting segment. + +The transcript contains discussion, competing preferences, and possible options +for the weekly dashboard. No participant approves an option, rejects an option +on behalf of the group, assigns a follow-up, agrees to gather more information, +or explicitly decides to defer the decision. + +Ground truth: + +- No decision was reached. +- No process decision was reached. +- No agreed next step was created. + +Typical LLM mistakes: + +- Treating a preference as a decision. +- Treating a proposed option as the selected option. +- Treating the topic change as an implicit deferral decision. +- Creating an action item for Dana even though she is only mentioned as absent. diff --git a/tests/gold/decision_none/expected.json b/tests/gold/decision_none/expected.json new file mode 100644 index 0000000..8131e21 --- /dev/null +++ b/tests/gold/decision_none/expected.json @@ -0,0 +1,24 @@ +{ + "facts": [], + "decisions": [], + "todos": [], + "questions": [], + "positions": [ + { + "speaker": "Jonas", + "position": "Jonas prefers moving the weekly dashboard to Monday morning.", + "evidence": "Jonas: I would prefer moving it to Monday morning." + }, + { + "speaker": "Mira", + "position": "Mira thinks Tuesday might work better for support.", + "evidence": "Mira: Tuesday might work better for support." + }, + { + "speaker": "Lea", + "position": "Lea suggests keeping Friday and making the dashboard shorter.", + "evidence": "Lea: Or we keep Friday and make the dashboard shorter." + } + ], + "technical": [] +} diff --git a/tests/gold/decision_none/transcript.txt b/tests/gold/decision_none/transcript.txt new file mode 100644 index 0000000..82fb7ec --- /dev/null +++ b/tests/gold/decision_none/transcript.txt @@ -0,0 +1,25 @@ +Mira: We need to talk about the weekly dashboard. + +Jonas: I would prefer moving it to Monday morning. + +Lea: Monday is rough for support because backlog cleanup starts then. + +Mira: Tuesday might work better for support. + +Jonas: Tuesday is hard for sales, at least this month. + +Lea: Or we keep Friday and make the dashboard shorter. + +Mira: I am not convinced shorter solves the timing issue. + +Jonas: I am not convinced Monday is actually a problem for everyone. + +Lea: Dana might have a view, but she is not here. + +Mira: We are circling now. + +Jonas: Yes, I do not have anything else to add. + +Lea: Same here. + +Mira: Okay, let's move to the budget topic. diff --git a/tests/gold/decision_simple/README.md b/tests/gold/decision_simple/README.md new file mode 100644 index 0000000..f5aa89d --- /dev/null +++ b/tests/gold/decision_simple/README.md @@ -0,0 +1,11 @@ +# decision_simple + +Tests one explicit decision with clear agreement language. + +The difficult part is separating the decision from nearby rationale about user confusion and from the non-decision statement that the copy can stay unchanged for now. + +Typical LLM mistakes: + +- Extracting the rationale as a separate decision. +- Treating "copy can stay as it is" as a formal decision. +- Losing the evidence that shows explicit agreement. diff --git a/tests/gold/decision_simple/expected.json b/tests/gold/decision_simple/expected.json new file mode 100644 index 0000000..1d97510 --- /dev/null +++ b/tests/gold/decision_simple/expected.json @@ -0,0 +1,13 @@ +{ + "facts": [], + "decisions": [ + { + "decision": "The welcome email will be sent after account activation.", + "evidence": "So are we agreed that the welcome email moves to after activation? Ben: Agreed. Cara: Yes, let's do that." + } + ], + "todos": [], + "questions": [], + "positions": [], + "technical": [] +} diff --git a/tests/gold/decision_simple/transcript.txt b/tests/gold/decision_simple/transcript.txt new file mode 100644 index 0000000..92860fc --- /dev/null +++ b/tests/gold/decision_simple/transcript.txt @@ -0,0 +1,19 @@ +Anna: Before we leave the onboarding flow, can we settle the email step? + +Ben: I still think the welcome email should go out after account activation, not before. + +Cara: Yes, before activation it keeps confusing people. + +Anna: So are we agreed that the welcome email moves to after activation? + +Ben: Agreed. + +Cara: Yes, let's do that. + +Anna: Good. Then that is the decision for this release. + +Ben: Separate note on the copy: I am not proposing any wording decision today. + +Cara: Same here, no wording proposal from me. + +Anna: Okay, then the only decision is the timing after activation. diff --git a/tests/gold/evil_meeting/README.md b/tests/gold/evil_meeting/README.md new file mode 100644 index 0000000..0a69519 --- /dev/null +++ b/tests/gold/evil_meeting/README.md @@ -0,0 +1,14 @@ +# evil_meeting + +Tests a deliberately difficult meeting with interruptions, corrections, topic switches, changed positions, absent referenced people, and near-decisions. + +Every utterance is designed to trigger a common extraction failure. The meeting mentions Omar and Platform, but neither is a participant. It includes an explicit non-decision on migration and a real decision only on excluding FR-7 from Friday's batch. + +Typical LLM mistakes: + +- Extracting a migration decision even though the group says no migration decision today. +- Assigning Dana a todo even though she retracts it. +- Treating Omar as a participant or technical owner. +- Claiming Platform approved something despite being absent. +- Losing the correction from "old export" to "nightly CSV job" and from API export to CSV export. +- Treating Dana's opinion about rollout appearance as a fact or decision. diff --git a/tests/gold/evil_meeting/expected.json b/tests/gold/evil_meeting/expected.json new file mode 100644 index 0000000..b6d6b69 --- /dev/null +++ b/tests/gold/evil_meeting/expected.json @@ -0,0 +1,73 @@ +{ + "facts": [ + { + "fact": "Tenant FR-7 still used the nightly CSV job yesterday.", + "evidence": "Alex: Fine. So fact: tenant FR-7 still used the nightly CSV job yesterday." + }, + { + "fact": "Omar is the customer contact, not the technical owner.", + "evidence": "Bea: Omar is the customer contact, not a participant here and not the technical owner." + }, + { + "fact": "Platform is the technical owner, but nobody from Platform is in the meeting.", + "evidence": "Chen: The technical owner is still Platform, but nobody from Platform is in this call." + }, + { + "fact": "Friday's rollout batch still includes DE-2 and NL-4.", + "evidence": "Alex: Good. Back to the portal rollout. Friday's batch still includes DE-2 and NL-4." + } + ], + "decisions": [ + { + "decision": "FR-7 is excluded from Friday's portal rollout batch.", + "evidence": "the portal rollout note will say FR-7 is excluded from Friday's batch. Bea: Agreed. Excluded from Friday's batch. Chen: Yes, put that in." + } + ], + "todos": [ + { + "task": "Check the FR-7 mapping table.", + "responsible": "Bea", + "deadline": "Thursday morning", + "evidence": "Bea: Yes, I will check it by Thursday morning." + } + ], + "questions": [ + { + "question": "Can FR-7 use the v2 mapping without a customer-side field rename?", + "evidence": "Alex: Open question: can FR-7 use the v2 mapping without a customer-side field rename?" + }, + { + "question": "Can Omar confirm FR-7's preferred launch window?", + "evidence": "Dana: Also, can Omar confirm their preferred launch window?" + } + ], + "positions": [ + { + "speaker": "Dana", + "position": "Dana wants to migrate FR-7 but recognizes the mapping table may not be clean.", + "evidence": "Dana: I want to, but we do not know if the mapping table is clean." + }, + { + "speaker": "Bea", + "position": "Bea changed her earlier position and now says not to migrate FR-7 until the mapping table is checked.", + "evidence": "I said last week we should migrate it. I am changing that. Do not migrate until the mapping table is checked." + }, + { + "speaker": "Dana", + "position": "Dana thinks excluding FR-7 makes the rollout look messy.", + "evidence": "Dana: I personally think excluding FR-7 makes the rollout look messy." + } + ], + "technical": [ + { + "subject": "French export failure", + "statement": "The old nightly CSV job failed; the new exporter was not running on tenant FR-7.", + "evidence": "The old export failed. The new exporter was not running on that tenant." + }, + { + "subject": "Export mapping tables", + "statement": "The API export uses the v2 mapping table, while the nightly CSV job uses the legacy table.", + "evidence": "the API export uses the v2 mapping table, the nightly CSV job uses the legacy table." + } + ] +} diff --git a/tests/gold/evil_meeting/transcript.txt b/tests/gold/evil_meeting/transcript.txt new file mode 100644 index 0000000..52fac0c --- /dev/null +++ b/tests/gold/evil_meeting/transcript.txt @@ -0,0 +1,61 @@ +Alex: Okay, quick pass on the portal rollout. Wait, before that, the French CSV export broke again. + +Bea: It did not break again. The old export failed. The new exporter was not running on that tenant. + +Chen: Sorry, when you say old export, do you mean the nightly job? + +Bea: Yes, the nightly CSV job. Not the API export. + +Alex: Fine. So fact: tenant FR-7 still used the nightly CSV job yesterday. + +Dana: I thought Omar owned that tenant. + +Bea: Omar is the customer contact, not a participant here and not the technical owner. + +Chen: The technical owner is still Platform, but nobody from Platform is in this call. + +Alex: Should we decide to migrate FR-7 today? + +Dana: I want to, but we do not know if the mapping table is clean. + +Bea: Also, I said last week we should migrate it. I am changing that. Do not migrate until the mapping table is checked. + +Chen: So no migration decision today? + +Alex: Correct, no migration decision today. + +Dana: But we can decide one thing: the portal rollout note will say FR-7 is excluded from Friday's batch. + +Bea: Agreed. Excluded from Friday's batch. + +Chen: Yes, put that in. + +Alex: Action item: Bea checks the FR-7 mapping table by Thursday morning. + +Bea: Yes, I will check it by Thursday morning. + +Dana: And I will message Omar after Bea is done. + +Alex: Hold on, after Bea is done is not a date. + +Dana: Fair. Then no task for me yet. I need Bea's result first. + +Chen: Technical note: the API export uses the v2 mapping table, the nightly CSV job uses the legacy table. + +Bea: Correct. + +Alex: Open question: can FR-7 use the v2 mapping without a customer-side field rename? + +Dana: Also, can Omar confirm their preferred launch window? + +Chen: Omar can answer that, but again he is not in this meeting. + +Alex: Good. Back to the portal rollout. Friday's batch still includes DE-2 and NL-4. + +Bea: Yes, those two are unchanged. + +Dana: I personally think excluding FR-7 makes the rollout look messy. + +Alex: Noted as Dana's view, not a decision. + +Chen: And please don't write that Platform approved anything. They are absent. diff --git a/tests/gold/facts_simple/README.md b/tests/gold/facts_simple/README.md new file mode 100644 index 0000000..fca4d5a --- /dev/null +++ b/tests/gold/facts_simple/README.md @@ -0,0 +1,11 @@ +# facts_simple + +Tests extraction of objective facts from a short status update. + +The transcript includes a question and a technical statement, but no decision or todo. + +Typical LLM mistakes: + +- Treating "Good" as approval of a decision. +- Turning "No decision needed today" into a decision. +- Missing that the scanner gateway details are technical as well as factual. diff --git a/tests/gold/facts_simple/expected.json b/tests/gold/facts_simple/expected.json new file mode 100644 index 0000000..751c8a7 --- /dev/null +++ b/tests/gold/facts_simple/expected.json @@ -0,0 +1,32 @@ +{ + "facts": [ + { + "fact": "The old scanner gateway is still running in aisle three.", + "evidence": "Tom: The old scanner gateway is still running in aisle three." + }, + { + "fact": "Aisles one and two moved to the new gateway last week.", + "evidence": "Tom: Yes. Aisles one and two moved to the new gateway last week." + }, + { + "fact": "The new gateway is handling live scans for receiving.", + "evidence": "Iris: The new gateway is already handling live scans for receiving." + } + ], + "decisions": [], + "todos": [], + "questions": [ + { + "question": "Is aisle three the only scanner gateway still left on the old gateway?", + "evidence": "Elena: Is that the only one left?" + } + ], + "positions": [], + "technical": [ + { + "subject": "Scanner gateway rollout", + "statement": "Aisle three remains on the old scanner gateway while aisles one and two use the new gateway.", + "evidence": "The old scanner gateway is still running in aisle three. Aisles one and two moved to the new gateway last week." + } + ] +} diff --git a/tests/gold/facts_simple/transcript.txt b/tests/gold/facts_simple/transcript.txt new file mode 100644 index 0000000..a5609a1 --- /dev/null +++ b/tests/gold/facts_simple/transcript.txt @@ -0,0 +1,19 @@ +Elena: Quick status on the warehouse migration. + +Tom: The old scanner gateway is still running in aisle three. + +Elena: Is that the only one left? + +Tom: Yes. Aisles one and two moved to the new gateway last week. + +Iris: The new gateway is already handling live scans for receiving. + +Elena: Good. Let's keep the rollout note factual. + +Tom: No decision needed today. + +Iris: Fine. + +Elena: Anything else on warehouse? + +Tom: No, that is all. diff --git a/tests/gold/facts_vs_positions/README.md b/tests/gold/facts_vs_positions/README.md new file mode 100644 index 0000000..4a6efdb --- /dev/null +++ b/tests/gold/facts_vs_positions/README.md @@ -0,0 +1,11 @@ +# facts_vs_positions + +Tests separation of objective facts from personal opinions. + +The transcript deliberately mixes numeric facts, named non-respondents, and subjective positions about rollout health. + +Typical LLM mistakes: + +- Treating Marta's opinion as an objective fact. +- Treating Leo's optimism as a fact. +- Extracting a rollout decision even though the group explicitly does not decide. diff --git a/tests/gold/facts_vs_positions/expected.json b/tests/gold/facts_vs_positions/expected.json new file mode 100644 index 0000000..15d490b --- /dev/null +++ b/tests/gold/facts_vs_positions/expected.json @@ -0,0 +1,42 @@ +{ + "facts": [ + { + "fact": "The pilot survey closed yesterday with 42 responses.", + "evidence": "Hannah: The pilot survey closed yesterday with 42 responses." + }, + { + "fact": "The average pilot survey rating was 3.8 out of 5.", + "evidence": "Leo: The average rating was 3.8 out of 5." + }, + { + "fact": "Northwind, Verdan, and Eastport did not respond to the survey.", + "evidence": "Hannah: That part is true. Northwind, Verdan, and Eastport did not respond." + } + ], + "decisions": [], + "todos": [], + "questions": [ + { + "question": "Why does Marta think the survey result is weaker than it looks?", + "evidence": "Leo: Why?" + } + ], + "positions": [ + { + "speaker": "Marta", + "position": "Marta thinks the pilot survey result is weaker than it looks.", + "evidence": "Marta: I think that is weaker than it looks." + }, + { + "speaker": "Leo", + "position": "Leo feels the pilot is healthy.", + "evidence": "Leo: I still feel the pilot is healthy." + }, + { + "speaker": "Marta", + "position": "Marta thinks the rollout should slow down.", + "evidence": "Marta: I disagree. My view is that we should slow down the rollout." + } + ], + "technical": [] +} diff --git a/tests/gold/facts_vs_positions/transcript.txt b/tests/gold/facts_vs_positions/transcript.txt new file mode 100644 index 0000000..967d289 --- /dev/null +++ b/tests/gold/facts_vs_positions/transcript.txt @@ -0,0 +1,19 @@ +Hannah: The pilot survey closed yesterday with 42 responses. + +Leo: The average rating was 3.8 out of 5. + +Marta: I think that is weaker than it looks. + +Leo: Why? + +Marta: Because three enterprise customers skipped the survey entirely. + +Hannah: That part is true. Northwind, Verdan, and Eastport did not respond. + +Leo: I still feel the pilot is healthy. + +Marta: I disagree. My view is that we should slow down the rollout. + +Hannah: Let's capture both views and not decide rollout speed today. + +Leo: Okay, that matches my notes. diff --git a/tests/gold/mixed_small/README.md b/tests/gold/mixed_small/README.md new file mode 100644 index 0000000..63d7cf3 --- /dev/null +++ b/tests/gold/mixed_small/README.md @@ -0,0 +1,11 @@ +# mixed_small + +Tests a compact realistic meeting containing all major extraction categories. + +The transcript includes facts, one explicit decision, one action item, one open question, one opinion, and a technical constraint. + +Typical LLM mistakes: + +- Applying the demo-only decision to production. +- Treating Mateo's diagnosis as a fact instead of a position. +- Creating a long-term normalizer decision even though it is explicitly open. diff --git a/tests/gold/mixed_small/expected.json b/tests/gold/mixed_small/expected.json new file mode 100644 index 0000000..4eb158e --- /dev/null +++ b/tests/gold/mixed_small/expected.json @@ -0,0 +1,50 @@ +{ + "facts": [ + { + "fact": "The staging import handled 12,000 rows last night.", + "evidence": "Priya: The staging import handled 12,000 rows last night." + }, + { + "fact": "The staging import took 48 minutes.", + "evidence": "Mateo: It finished, but it took 48 minutes." + }, + { + "fact": "The partner demo target is 30 minutes.", + "evidence": "Lena: Yes, that is still the demo target." + } + ], + "decisions": [ + { + "decision": "Address normalization will be disabled for the demo import only.", + "evidence": "Can we agree to disable address normalization for the demo import only? Lena: Yes, for the demo import only. Mateo: Agreed." + } + ], + "todos": [ + { + "task": "Update the demo import configuration.", + "responsible": "Mateo", + "deadline": "Friday noon", + "evidence": "Mateo: I will do that before Friday noon." + } + ], + "questions": [ + { + "question": "Whether a faster normalizer is needed after the demo.", + "evidence": "Lena: And the open question is whether we need a faster normalizer after the demo." + } + ], + "positions": [ + { + "speaker": "Mateo", + "position": "Mateo thinks address normalization is the slow part.", + "evidence": "Mateo: I think the slow part is address normalization." + } + ], + "technical": [ + { + "subject": "Demo import configuration", + "statement": "Address normalization is disabled only for the demo import; production imports keep full normalization.", + "evidence": "for the demo import only. Production imports keep the full normalization." + } + ] +} diff --git a/tests/gold/mixed_small/transcript.txt b/tests/gold/mixed_small/transcript.txt new file mode 100644 index 0000000..cb282ce --- /dev/null +++ b/tests/gold/mixed_small/transcript.txt @@ -0,0 +1,23 @@ +Priya: The staging import handled 12,000 rows last night. + +Mateo: It finished, but it took 48 minutes. + +Priya: The limit for the partner demo is 30 minutes, right? + +Lena: Yes, that is still the demo target. + +Mateo: I think the slow part is address normalization. + +Priya: Can we agree to disable address normalization for the demo import only? + +Lena: Yes, for the demo import only. + +Mateo: Agreed. Production imports keep the full normalization. + +Priya: Mateo, please update the demo config before Friday noon. + +Mateo: I will do that before Friday noon. + +Lena: And the open question is whether we need a faster normalizer after the demo. + +Priya: Capture that, but no decision on the long-term fix today. diff --git a/tests/gold/question_simple/README.md b/tests/gold/question_simple/README.md new file mode 100644 index 0000000..dd1b5aa --- /dev/null +++ b/tests/gold/question_simple/README.md @@ -0,0 +1,11 @@ +# question_simple + +Tests extraction of an explicit open question. + +The transcript also contains a non-task: Kai says he can ask finance but explicitly does not accept it as a task yet. + +Typical LLM mistakes: + +- Creating a todo for Kai despite his correction. +- Missing that the group decides to leave the issue open. +- Treating "support package" as enough information to answer the question. diff --git a/tests/gold/question_simple/expected.json b/tests/gold/question_simple/expected.json new file mode 100644 index 0000000..cdcd420 --- /dev/null +++ b/tests/gold/question_simple/expected.json @@ -0,0 +1,27 @@ +{ + "facts": [ + { + "fact": "The vendor invoice came in this morning.", + "evidence": "Kai: The vendor invoice came in this morning." + }, + { + "fact": "The invoice line item says support package.", + "evidence": "Kai: I don't know. The line item just says support package." + } + ], + "decisions": [ + { + "decision": "The support-hours invoice issue will remain an open question for now.", + "evidence": "Ruth: Fine. Let's leave it as an open question for now." + } + ], + "todos": [], + "questions": [ + { + "question": "Does the vendor invoice include the extra support hours from March?", + "evidence": "Ruth: Does it include the extra support hours from March?" + } + ], + "positions": [], + "technical": [] +} diff --git a/tests/gold/question_simple/transcript.txt b/tests/gold/question_simple/transcript.txt new file mode 100644 index 0000000..ba13551 --- /dev/null +++ b/tests/gold/question_simple/transcript.txt @@ -0,0 +1,19 @@ +Kai: The vendor invoice came in this morning. + +Ruth: Does it include the extra support hours from March? + +Kai: I don't know. The line item just says support package. + +Ruth: Then that is still open. + +Kai: I can ask finance, but I am not taking that as a task yet. + +Ruth: Fine. Let's leave it as an open question for now. + +Kai: Understood. + +Ruth: Anything else on invoices? + +Kai: No. + +Ruth: Then next item. diff --git a/tests/gold/technical_simple/README.md b/tests/gold/technical_simple/README.md new file mode 100644 index 0000000..9c21e37 --- /dev/null +++ b/tests/gold/technical_simple/README.md @@ -0,0 +1,11 @@ +# technical_simple + +Tests technical extraction with a corrected diagnosis. + +The transcript contrasts two possible causes: certificate expiry and runner configuration. The latter is confirmed as the cause. + +Typical LLM mistakes: + +- Reporting certificate expiry as the problem. +- Creating a fix decision even though the group explicitly says no fix is decided. +- Missing the distinction between fact and technical diagnosis. diff --git a/tests/gold/technical_simple/expected.json b/tests/gold/technical_simple/expected.json new file mode 100644 index 0000000..bd16c7b --- /dev/null +++ b/tests/gold/technical_simple/expected.json @@ -0,0 +1,33 @@ +{ + "facts": [ + { + "fact": "The mobile build failed on the staging runner.", + "evidence": "Sofia: The mobile build failed again on the staging runner." + }, + { + "fact": "The certificate is valid until October.", + "evidence": "Nils: The certificate itself is valid until October." + } + ], + "decisions": [], + "todos": [], + "questions": [ + { + "question": "Is the mobile build failing with the same error as yesterday?", + "evidence": "Nils: Same error as yesterday?" + } + ], + "positions": [], + "technical": [ + { + "subject": "iOS staging build", + "statement": "The iOS job fails during code signing because the runner uses the old keychain path.", + "evidence": "The iOS job now fails during code signing. The problem is that the runner uses the old keychain path." + }, + { + "subject": "Failure classification", + "statement": "The failure is a runner configuration issue, not a certificate expiry issue.", + "evidence": "So it is a runner configuration issue, not a certificate expiry issue. Sofia: Exactly." + } + ] +} diff --git a/tests/gold/technical_simple/transcript.txt b/tests/gold/technical_simple/transcript.txt new file mode 100644 index 0000000..5aa7076 --- /dev/null +++ b/tests/gold/technical_simple/transcript.txt @@ -0,0 +1,19 @@ +Sofia: The mobile build failed again on the staging runner. + +Nils: Same error as yesterday? + +Sofia: No, different. The iOS job now fails during code signing. + +Nils: The certificate itself is valid until October. + +Sofia: Right. The problem is that the runner uses the old keychain path. + +Nils: So it is a runner configuration issue, not a certificate expiry issue. + +Sofia: Exactly. + +Nils: We are not deciding the fix today. + +Sofia: Okay. + +Nils: Next item. diff --git a/tests/gold/todo_negative/README.md b/tests/gold/todo_negative/README.md new file mode 100644 index 0000000..f9d5580 --- /dev/null +++ b/tests/gold/todo_negative/README.md @@ -0,0 +1,11 @@ +# todo_negative + +Tests that vague "someone should" language is not an action item. + +The transcript contains a real decision to keep the issue on the risk list, but no assigned task. + +Typical LLM mistakes: + +- Creating a todo with "someone" as owner. +- Assigning Paula, Ravi, or Sam even though they explicitly do not take ownership. +- Ignoring the explicit "no owner for now" correction. diff --git a/tests/gold/todo_negative/expected.json b/tests/gold/todo_negative/expected.json new file mode 100644 index 0000000..e49e897 --- /dev/null +++ b/tests/gold/todo_negative/expected.json @@ -0,0 +1,22 @@ +{ + "facts": [ + { + "fact": "The customer export takes longer than Paula expected.", + "evidence": "Paula: The customer export takes longer than I expected." + }, + { + "fact": "There is no owner for the customer export issue for now.", + "evidence": "Paula: Right, no owner for now." + } + ], + "decisions": [ + { + "decision": "The customer export issue will stay on the risk list for now.", + "evidence": "Sam: Then let's just keep it on the risk list. Paula: Right, no owner for now." + } + ], + "todos": [], + "questions": [], + "positions": [], + "technical": [] +} diff --git a/tests/gold/todo_negative/transcript.txt b/tests/gold/todo_negative/transcript.txt new file mode 100644 index 0000000..8430b43 --- /dev/null +++ b/tests/gold/todo_negative/transcript.txt @@ -0,0 +1,19 @@ +Paula: The customer export takes longer than I expected. + +Ravi: Someone should probably look at it. + +Sam: Yes, maybe after the release freeze. + +Paula: I don't have capacity this week. + +Ravi: Same here. + +Sam: Then let's just keep it on the risk list. + +Paula: Right, no owner for now. + +Ravi: We can revisit it in planning. + +Sam: Okay. + +Paula: Next item. diff --git a/tests/gold/todo_simple/README.md b/tests/gold/todo_simple/README.md new file mode 100644 index 0000000..fe833c7 --- /dev/null +++ b/tests/gold/todo_simple/README.md @@ -0,0 +1,11 @@ +# todo_simple + +Tests one clear action item with responsible person and deadline. + +The difficult part is not adding extra scope: Nina explicitly says she will not touch the layout. + +Typical LLM mistakes: + +- Adding a layout update as a task. +- Dropping the deadline. +- Turning Omar's request into the task evidence instead of Nina's commitment. diff --git a/tests/gold/todo_simple/expected.json b/tests/gold/todo_simple/expected.json new file mode 100644 index 0000000..f3b97c6 --- /dev/null +++ b/tests/gold/todo_simple/expected.json @@ -0,0 +1,20 @@ +{ + "facts": [ + { + "fact": "The beta signup page points to the old privacy note.", + "evidence": "Omar: The beta signup page still points to the old privacy note." + } + ], + "decisions": [], + "todos": [ + { + "task": "Update the privacy link on the beta signup page.", + "responsible": "Nina", + "deadline": "Thursday noon", + "evidence": "Nina: Yes, I will update the privacy link by Thursday noon." + } + ], + "questions": [], + "positions": [], + "technical": [] +} diff --git a/tests/gold/todo_simple/transcript.txt b/tests/gold/todo_simple/transcript.txt new file mode 100644 index 0000000..57c0382 --- /dev/null +++ b/tests/gold/todo_simple/transcript.txt @@ -0,0 +1,19 @@ +Omar: The beta signup page still points to the old privacy note. + +Nina: Yes, I saw that yesterday. + +Omar: Can you update the link before the partner demo? + +Nina: Yes, I will update the privacy link by Thursday noon. + +Omar: Great. Nothing else on that page from my side. + +Nina: I will only touch the link, not the layout. + +Omar: Fine. + +Nina: Then I have what I need. + +Omar: We can move on. + +Nina: Yes. diff --git a/tests/test_extraction_protocol.py b/tests/test_extraction_protocol.py index dbadf9d..412f1c8 100644 --- a/tests/test_extraction_protocol.py +++ b/tests/test_extraction_protocol.py @@ -5,6 +5,7 @@ from pathlib import Path from src.meeting_lab.extraction.extract_chunks import ( EXTRACTION_CATEGORIES, + build_prompt, extraction_path_for_chunk, normalize_current_schema, parse_json_response, @@ -93,6 +94,14 @@ Final answer: self.assertEqual(parsed["facts"], ["final"]) self.assertEqual(set(parsed), set(EXTRACTION_CATEGORIES)) + def test_build_prompt_includes_common_and_decision_prompt_files(self) -> None: + prompt = build_prompt("transcript.txt", "Anna: Agreed.") + + self.assertIn("Du extrahierst Informationen aus Meeting-Transkripten.", prompt) + self.assertIn("You extract decisions from meeting transcript text.", prompt) + self.assertIn("Extract each decision as one atomic commitment.", prompt) + self.assertIn("Anna: Agreed.", prompt) + def test_build_protocol_groups_extraction_items(self) -> None: with tempfile.TemporaryDirectory() as directory: input_dir = Path(directory) diff --git a/tests/test_gold_runner.py b/tests/test_gold_runner.py new file mode 100644 index 0000000..bc522d4 --- /dev/null +++ b/tests/test_gold_runner.py @@ -0,0 +1,51 @@ +import json +import tempfile +import unittest +from pathlib import Path + +from scripts.run_gold_test import read_json_object, scenario_paths, validate_required_keys + + +class GoldRunnerTests(unittest.TestCase): + def test_missing_transcript_is_reported(self) -> None: + with tempfile.TemporaryDirectory() as directory: + scenario = Path(directory) + (scenario / "expected.json").write_text("{}", encoding="utf-8") + + with self.assertRaisesRegex(FileNotFoundError, "Missing transcript.txt"): + scenario_paths(scenario) + + def test_missing_expected_is_reported(self) -> None: + with tempfile.TemporaryDirectory() as directory: + scenario = Path(directory) + (scenario / "transcript.txt").write_text("A: Hello.", encoding="utf-8") + + with self.assertRaisesRegex(FileNotFoundError, "Missing expected.json"): + scenario_paths(scenario) + + def test_invalid_expected_json_is_reported(self) -> None: + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "expected.json" + path.write_text("{invalid", encoding="utf-8") + + with self.assertRaisesRegex(ValueError, "Invalid JSON"): + read_json_object(path) + + def test_missing_required_schema_keys_are_reported(self) -> None: + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "expected.json" + data = { + "facts": [], + "decisions": [], + "todos": [], + "questions": [], + "positions": [], + } + path.write_text(json.dumps(data), encoding="utf-8") + + with self.assertRaisesRegex(ValueError, "technical"): + validate_required_keys(read_json_object(path), path) + + +if __name__ == "__main__": + unittest.main()