3 Commits
Author SHA1 Message Date
admin f7ad9ba51f Establish prompt engineering baseline with Gold Standard tests
- introduce Gold Standard evaluation corpus
- document decision taxonomy
- define prompt-engineering methodology
- add regression workflow
- establish Prompt Version 2 baseline
- validate decision_simple, decision_deferred and decision_none
2026-07-30 12:13:10 +02:00
admin 07b0d80113 Implement first end-to-end meeting analysis pipeline 2026-07-30 09:15:38 +02:00
admin 46565233d8 Fix Whisper JSON chunk extraction 2026-07-29 15:34:01 +02:00
48 changed files with 2912 additions and 0 deletions
+8
View File
@@ -32,6 +32,14 @@ htmlcov/
experiments/**/output/
experiments/**/results/
# Pipeline runtime artifacts
samples/raw/
samples/whisper/
**/chunk_*_extraction.json
**/meeting_protocol.md
**/*.raw.txt
tests/gold/**/actual.json
# Lokale Meetings (niemals versionieren)
meeting_data/
recordings/
+26
View File
@@ -0,0 +1,26 @@
Du extrahierst Informationen aus Meeting-Transkripten.
Arbeite ausschließlich mit dem vorgelegten Transkript.
Verwende kein eigenes Fachwissen, keine Vermutungen und keine üblichen
Funktionsweisen technischer Systeme.
Regeln:
1. Erfinde nichts.
2. Interpretiere technische Aussagen nicht über den Wortlaut hinaus.
3. Korrigiere keine Aussagen anhand vermeintlichen Weltwissens.
4. Wenn etwas widersprüchlich oder unklar ist, kennzeichne es als unklar.
5. Übernimm wichtige technische Aussagen möglichst nah am Wortlaut.
6. Nenne bei Fakten nach Möglichkeit den Sprecher.
7. Ein Beschluss ist nur dann ein Beschluss, wenn im Text eine Einigung,
Freigabe oder verbindliche Festlegung erkennbar ist.
8. Eine Aufgabe ist nur dann eine Aufgabe, wenn eine Handlung und möglichst
eine verantwortliche Person oder Organisation erkennbar sind.
9. Gib ausschließlich gültiges JSON aus. Kein Markdown, keine Erläuterungen.
Hinweise zur Ausgabe:
- Alle obersten Schlüssel müssen vorhanden sein.
- Verwende leere Listen, wenn keine Einträge vorhanden sind.
- Verwende null, wenn Verantwortliche, Sprecher oder Termine nicht erkennbar sind.
- "evidence" muss sich eng am Transkript orientieren.
- Ersetze technische Aussagen niemals durch eine vermeintlich korrektere Erklärung.
- Confidence-Werte sind ausdrücklich nicht erwünscht.
+86
View File
@@ -0,0 +1,86 @@
You extract decisions from meeting transcript text.
Return only valid JSON.
Schema:
{
"decisions": [
{
"decision": "Concise decision text",
"evidence": "Short quote from the transcript"
}
]
}
Definition:
A decision exists only when the participants explicitly agreed, approved,
confirmed, adopted, assigned, or otherwise made something binding during the
meeting.
Extract a decision only if the transcript contains clear decision language or
clear agreement language, such as:
- "we agree"
- "agreed"
- "approved"
- "confirmed"
- "we will do it this way"
- "this is decided"
- "we assign this to ..."
- "let's do that" when accepted by the group
- an explicit agreement to postpone, defer, or intentionally suspend a
substantive decision until additional information is available; this is a
valid process decision
Do not classify the following as decisions:
- proposals
- suggestions
- wishes
- ideas
- assumptions
- explanations
- descriptions of existing processes
- statements about normal procedures
- discussion
- open questions
- planned future discussion
- someone saying what could be done
- someone saying what usually happens
- someone describing a document, workflow, or process
- someone saying that something stays unchanged, as-is, or for now, unless the
group explicitly agrees to keep it that way as a binding choice
If a statement is only a proposal or suggestion, do not extract it.
If participants discuss something but do not explicitly agree to it, do not
extract it.
If the transcript describes an existing process, rule, template, document, or
workflow, do not extract it unless the participants explicitly adopt or change it
in this meeting.
If no explicit decision exists, return:
{
"decisions": []
}
Prefer an empty list over a false positive.
For each decision:
- Write one concise decision text.
- Extract each decision as one atomic commitment.
- Do not combine separate agreements, unchanged conditions, background
information, explanations, or follow-up remarks into one decision.
- If two distinct matters were agreed, return two separate decisions.
- Include one short evidence quote from the transcript.
- Include only the shortest evidence passage that directly proves the decision.
- Do not invent responsible persons.
- Do not invent deadlines.
- Do not invent priorities.
- Do not add confidence values.
- Do not explain your reasoning.
+139
View File
@@ -0,0 +1,139 @@
import argparse
import json
import math
from pathlib import Path
from typing import Any
def is_nonfinite_number(value: Any) -> bool:
"""Prüft auf NaN sowie positive oder negative Unendlichkeit."""
return isinstance(value, float) and not math.isfinite(value)
def sanitize_nonfinite_values(value: Any) -> Any:
"""
Ersetzt NaN und Infinity rekursiv durch None.
None wird in JSON als null geschrieben.
"""
if is_nonfinite_number(value):
return None
if isinstance(value, dict):
return {
key: sanitize_nonfinite_values(item)
for key, item in value.items()
}
if isinstance(value, list):
return [
sanitize_nonfinite_values(item)
for item in value
]
return value
def is_invalid_empty_segment(segment: dict[str, Any]) -> bool:
"""Erkennt technisch leere Whisper-Artefakte."""
text = str(segment.get("text", "")).strip()
start = segment.get("start")
end = segment.get("end")
avg_logprob = segment.get("avg_logprob")
logprob_is_nan = (
isinstance(avg_logprob, float)
and math.isnan(avg_logprob)
)
return (
not text
and start == end
and logprob_is_nan
)
def clean_whisper_json(
input_path: Path,
output_path: Path,
) -> None:
with input_path.open("r", encoding="utf-8") as file:
data = json.load(file)
original_segments = data.get("segments", [])
cleaned_segments = [
segment
for segment in original_segments
if not is_invalid_empty_segment(segment)
]
for new_id, segment in enumerate(cleaned_segments):
segment["id"] = new_id
data["segments"] = cleaned_segments
data["text"] = " ".join(
str(segment.get("text", "")).strip()
for segment in cleaned_segments
if str(segment.get("text", "")).strip()
)
# Alle noch vorhandenen NaN-/Infinity-Werte durch null ersetzen.
data = sanitize_nonfinite_values(data)
# Erst vollständig in einen String serialisieren.
# Dadurch bleibt keine unvollständige Ausgabedatei zurück,
# falls doch noch ein Fehler auftritt.
json_content = json.dumps(
data,
ensure_ascii=False,
indent=2,
allow_nan=False,
)
output_path.write_text(
json_content + "\n",
encoding="utf-8",
)
removed = len(original_segments) - len(cleaned_segments)
print(f"Eingabedatei: {input_path}")
print(f"Ausgabedatei: {output_path}")
print(f"Segmente vorher: {len(original_segments)}")
print(f"Segmente nachher: {len(cleaned_segments)}")
print(f"Segmente entfernt: {removed}")
def main() -> None:
parser = argparse.ArgumentParser(
description=(
"Entfernt technisch leere Artefakte aus "
"einer MLX-Whisper-JSON-Datei."
)
)
parser.add_argument(
"input",
type=Path,
help="Whisper-JSON-Datei",
)
parser.add_argument(
"-o",
"--output",
type=Path,
help="Ausgabedatei",
)
args = parser.parse_args()
output_path = args.output or args.input.with_name(
f"{args.input.stem}_cleaned.json"
)
clean_whisper_json(args.input, output_path)
if __name__ == "__main__":
main()
+223
View File
@@ -0,0 +1,223 @@
#!/usr/bin/env python3
"""Run one gold-corpus scenario through the existing extraction flow."""
from __future__ import annotations
import argparse
import json
import sys
import time
from pathlib import Path
from typing import Any
import requests
REPO_ROOT = Path(__file__).resolve().parents[1]
if str(REPO_ROOT) not in sys.path:
sys.path.insert(0, str(REPO_ROOT))
from src.meeting_lab.extraction.extract_chunks import (
DEFAULT_ENDPOINT,
EXTRACTION_CATEGORIES,
build_prompt,
call_ollama,
normalize_current_schema,
parse_json_response,
)
REQUIRED_KEYS = set(EXTRACTION_CATEGORIES)
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Run one gold-standard transcript through extraction."
)
parser.add_argument("scenario", type=Path, help="Gold scenario directory")
parser.add_argument("--model", required=True, help="Ollama model name")
parser.add_argument(
"--endpoint",
default=DEFAULT_ENDPOINT,
help=f"Ollama generate endpoint (default: {DEFAULT_ENDPOINT})",
)
parser.add_argument(
"--timeout",
type=int,
default=1800,
help="HTTP timeout in seconds (default: 1800)",
)
parser.add_argument(
"--temperature",
type=float,
default=0.0,
help="Sampling temperature (default: 0.0)",
)
parser.add_argument(
"--num-predict",
type=int,
default=8192,
help="Maximum generated tokens (default: 8192)",
)
parser.add_argument(
"--num-ctx",
type=int,
default=32768,
help="Context window tokens (default: 32768)",
)
return parser.parse_args()
def scenario_paths(scenario_dir: Path) -> tuple[Path, Path, Path]:
if not scenario_dir.is_dir():
raise FileNotFoundError(f"Scenario directory not found: {scenario_dir}")
transcript_path = scenario_dir / "transcript.txt"
expected_path = scenario_dir / "expected.json"
actual_path = scenario_dir / "actual.json"
if not transcript_path.is_file():
raise FileNotFoundError(f"Missing transcript.txt: {transcript_path}")
if not expected_path.is_file():
raise FileNotFoundError(f"Missing expected.json: {expected_path}")
return transcript_path, expected_path, actual_path
def read_json_object(path: Path) -> dict[str, Any]:
try:
data = json.loads(path.read_text(encoding="utf-8"))
except json.JSONDecodeError as exc:
raise ValueError(f"Invalid JSON in {path}: {exc}") from exc
if not isinstance(data, dict):
raise ValueError(f"JSON file must contain an object: {path}")
return data
def validate_required_keys(data: dict[str, Any], path: Path) -> None:
missing = sorted(REQUIRED_KEYS - set(data))
if missing:
raise ValueError(f"Missing required keys in {path}: {', '.join(missing)}")
def format_items(items: Any) -> list[str]:
if not isinstance(items, list):
return [f"<invalid non-list value: {items!r}>"]
if not items:
return ["<none>"]
return [str(item) for item in items]
def print_decision_comparison(
scenario_dir: Path,
model: str,
expected: dict[str, Any],
actual: dict[str, Any],
runtime: float,
) -> None:
expected_decisions = expected.get("decisions", [])
actual_decisions = actual.get("decisions", [])
print(f"Scenario: {scenario_dir}")
print(f"Model: {model}")
print(f"Expected decision count: {len(expected_decisions)}")
print(f"Actual decision count: {len(actual_decisions)}")
print("Expected decisions:")
for item in format_items(expected_decisions):
print(f"- {item}")
print("Actual decisions:")
for item in format_items(actual_decisions):
print(f"- {item}")
print(f"Runtime: {runtime:.2f}s")
def run_gold_test(
scenario_dir: Path,
model: str,
endpoint: str,
timeout: int,
temperature: float,
num_predict: int | None,
num_ctx: int | None,
) -> Path:
transcript_path, expected_path, actual_path = scenario_paths(scenario_dir)
expected = read_json_object(expected_path)
validate_required_keys(expected, expected_path)
transcript = transcript_path.read_text(encoding="utf-8-sig").strip()
if not transcript:
raise ValueError(f"The transcript is empty: {transcript_path}")
prompt = build_prompt(transcript_path.name, transcript)
started = time.perf_counter()
raw_text, _metadata = call_ollama(
endpoint=endpoint,
model=model,
prompt=prompt,
timeout=timeout,
temperature=temperature,
num_predict=num_predict,
num_ctx=num_ctx,
)
try:
parsed = parse_json_response(raw_text)
except (json.JSONDecodeError, ValueError) as exc:
raw_path = actual_path.with_suffix(".raw.txt")
raw_path.write_text(raw_text + "\n", encoding="utf-8")
raise ValueError(f"Model output was not valid JSON. Raw output: {raw_path}") from exc
actual = normalize_current_schema(parsed)
actual_path.write_text(
json.dumps(actual, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
actual_from_disk = read_json_object(actual_path)
validate_required_keys(actual_from_disk, actual_path)
runtime = time.perf_counter() - started
print_decision_comparison(
scenario_dir=scenario_dir,
model=model,
expected=expected,
actual=actual_from_disk,
runtime=runtime,
)
return actual_path
def main() -> int:
args = parse_args()
try:
actual_path = run_gold_test(
scenario_dir=args.scenario,
model=args.model,
endpoint=args.endpoint,
timeout=args.timeout,
temperature=args.temperature,
num_predict=args.num_predict,
num_ctx=args.num_ctx,
)
except requests.ConnectionError:
print(
"Error: Ollama is not reachable. Is `ollama serve` running?",
file=sys.stderr,
)
return 1
except requests.Timeout:
print("Error: The Ollama request timed out.", file=sys.stderr)
return 1
except requests.HTTPError as exc:
print(f"Error: Ollama returned an HTTP error: {exc}", file=sys.stderr)
return 1
except (OSError, UnicodeError, ValueError) as exc:
print(f"Error: {exc}", file=sys.stderr)
return 1
print(f"Actual JSON: {actual_path}")
return 0
if __name__ == "__main__":
raise SystemExit(main())
@@ -0,0 +1,325 @@
#!/usr/bin/env python3
"""
Split a meeting transcript into reasonably sized chunks without cutting
through transcript blocks.
The script treats paragraphs separated by blank lines as atomic blocks.
It tries to cut close to --target-chars and will not exceed --max-chars
unless a single block is already larger than that limit.
Example:
python chunk_transcript.py meeting.txt
python chunk_transcript.py meeting.txt --target-chars 9000 --max-chars 11000
python chunk_transcript.py meeting.txt --overlap-blocks 1
"""
from __future__ import annotations
import argparse
import json
import re
import sys
from dataclasses import asdict, dataclass
from pathlib import Path
from typing import Any
TIMESTAMP_RE = re.compile(
r"(?m)^\s*(?:"
r"\[(?:\d{1,2}:)?\d{1,2}:\d{2}\]"
r"|(?:\d{1,2}:)?\d{1,2}:\d{2}\s*(?:-->|-)"
r")"
)
@dataclass(frozen=True)
class ChunkInfo:
number: int
filename: str
chars: int
blocks: int
first_timestamp: str | None
last_timestamp: str | None
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Split a transcript into block-aligned text chunks."
)
parser.add_argument(
"input_file",
type=Path,
help="Transcript text file or Whisper JSON file",
)
parser.add_argument(
"-o",
"--output-dir",
type=Path,
help="Output directory; default: <input-stem>_chunks",
)
parser.add_argument(
"--target-chars",
type=int,
default=9000,
help="Preferred chunk size in characters (default: 9000)",
)
parser.add_argument(
"--max-chars",
type=int,
default=11000,
help="Soft maximum chunk size in characters (default: 11000)",
)
parser.add_argument(
"--min-chars",
type=int,
default=5000,
help="Preferred minimum before a chunk may be closed (default: 5000)",
)
parser.add_argument(
"--overlap-blocks",
type=int,
default=0,
help="Repeat this many trailing blocks in the next chunk (default: 0)",
)
return parser.parse_args()
def validate_args(args: argparse.Namespace) -> None:
if args.target_chars <= 0 or args.max_chars <= 0 or args.min_chars < 0:
raise ValueError("Character limits must be positive.")
if args.min_chars > args.target_chars:
raise ValueError("--min-chars must not exceed --target-chars.")
if args.target_chars > args.max_chars:
raise ValueError("--target-chars must not exceed --max-chars.")
if args.overlap_blocks < 0:
raise ValueError("--overlap-blocks must not be negative.")
def normalize_text(text: str) -> str:
text = text.replace("\r\n", "\n").replace("\r", "\n")
return text.strip()
def blocks_from_whisper_json(data: Any) -> list[str]:
if not isinstance(data, dict):
raise ValueError("Whisper JSON input must contain a JSON object.")
segments = data.get("segments")
if not isinstance(segments, list):
text = data.get("text")
if isinstance(text, str) and text.strip():
return split_into_blocks(normalize_text(text))
raise ValueError("Whisper JSON input must contain 'segments' or 'text'.")
blocks: list[str] = []
for segment in segments:
if not isinstance(segment, dict):
continue
text = str(segment.get("text", "")).strip()
if text:
blocks.append(text)
if not blocks:
raise ValueError("Whisper JSON input contains no transcript text.")
return blocks
def read_transcript_blocks(input_file: Path) -> list[str]:
source = normalize_text(input_file.read_text(encoding="utf-8-sig"))
if not source:
raise ValueError("The transcript is empty.")
if input_file.suffix.lower() == ".json":
return blocks_from_whisper_json(json.loads(source))
return split_into_blocks(source)
def split_into_blocks(text: str) -> list[str]:
"""
Prefer blank-line-delimited transcript blocks.
If the file has no blank lines but contains line-start timestamps,
split before each timestamp. Otherwise, use non-empty lines as blocks.
"""
paragraphs = [part.strip() for part in re.split(r"\n\s*\n+", text) if part.strip()]
if len(paragraphs) > 1:
return paragraphs
timestamp_starts = list(TIMESTAMP_RE.finditer(text))
if len(timestamp_starts) > 1:
blocks: list[str] = []
for index, match in enumerate(timestamp_starts):
start = match.start()
end = (
timestamp_starts[index + 1].start()
if index + 1 < len(timestamp_starts)
else len(text)
)
block = text[start:end].strip()
if block:
blocks.append(block)
prefix = text[: timestamp_starts[0].start()].strip()
if prefix:
blocks.insert(0, prefix)
return blocks
return [line.strip() for line in text.splitlines() if line.strip()]
def rendered_length(blocks: list[str]) -> int:
if not blocks:
return 0
return sum(len(block) for block in blocks) + 2 * (len(blocks) - 1)
def build_chunks(
blocks: list[str],
target_chars: int,
max_chars: int,
min_chars: int,
overlap_blocks: int,
) -> list[list[str]]:
chunks: list[list[str]] = []
current: list[str] = []
for block in blocks:
prospective = current + [block]
prospective_len = rendered_length(prospective)
current_len = rendered_length(current)
should_close = (
bool(current)
and current_len >= min_chars
and (
prospective_len > max_chars
or (current_len >= target_chars and prospective_len > target_chars)
)
)
if should_close:
chunks.append(current)
overlap = current[-overlap_blocks:] if overlap_blocks else []
current = overlap + [block]
else:
current.append(block)
if current:
chunks.append(current)
return rebalance_last_chunk(chunks, min_chars, max_chars)
def rebalance_last_chunk(
chunks: list[list[str]], min_chars: int, max_chars: int
) -> list[list[str]]:
"""
Avoid a tiny final chunk by moving complete blocks from the previous chunk.
"""
if len(chunks) < 2 or rendered_length(chunks[-1]) >= min_chars:
return chunks
previous = chunks[-2]
final = chunks[-1]
while len(previous) > 1 and rendered_length(final) < min_chars:
candidate = previous[-1]
new_final = [candidate] + final
if rendered_length(new_final) > max_chars:
break
final.insert(0, previous.pop())
return chunks
def extract_timestamps(text: str) -> list[str]:
return [match.group(0).strip() for match in TIMESTAMP_RE.finditer(text)]
def write_chunks(
chunks: list[list[str]], output_dir: Path, input_name: str
) -> list[ChunkInfo]:
output_dir.mkdir(parents=True, exist_ok=True)
infos: list[ChunkInfo] = []
width = max(2, len(str(len(chunks))))
for number, blocks in enumerate(chunks, start=1):
content = "\n\n".join(blocks).strip() + "\n"
filename = f"chunk_{number:0{width}d}.txt"
path = output_dir / filename
path.write_text(content, encoding="utf-8")
timestamps = extract_timestamps(content)
infos.append(
ChunkInfo(
number=number,
filename=filename,
chars=len(content),
blocks=len(blocks),
first_timestamp=timestamps[0] if timestamps else None,
last_timestamp=timestamps[-1] if timestamps else None,
)
)
manifest = {
"source_file": input_name,
"chunk_count": len(infos),
"chunks": [asdict(info) for info in infos],
}
(output_dir / "manifest.json").write_text(
json.dumps(manifest, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
return infos
def main() -> int:
args = parse_args()
try:
validate_args(args)
if not args.input_file.is_file():
raise FileNotFoundError(f"Input file not found: {args.input_file}")
blocks = read_transcript_blocks(args.input_file)
chunks = build_chunks(
blocks=blocks,
target_chars=args.target_chars,
max_chars=args.max_chars,
min_chars=args.min_chars,
overlap_blocks=args.overlap_blocks,
)
output_dir = args.output_dir or args.input_file.with_name(
f"{args.input_file.stem}_chunks"
)
infos = write_chunks(chunks, output_dir, args.input_file.name)
except (OSError, UnicodeError, ValueError) as exc:
print(f"Error: {exc}", file=sys.stderr)
return 1
print(f"Source: {args.input_file}")
print(f"Blocks: {len(blocks)}")
print(f"Chunks: {len(infos)}")
print(f"Output: {output_dir}")
print()
for info in infos:
time_range = ""
if info.first_timestamp or info.last_timestamp:
time_range = (
f" | {info.first_timestamp or '?'} to {info.last_timestamp or '?'}"
)
print(
f"{info.filename}: {info.chars:>6} chars, "
f"{info.blocks:>3} blocks{time_range}"
)
return 0
if __name__ == "__main__":
raise SystemExit(main())
@@ -0,0 +1,508 @@
#!/usr/bin/env python3
"""
Extract structured meeting information from normalized transcript chunks.
This module is adapted from /Users/tazl/quick-whisper-test/summarize.py.
It keeps the same Ollama extraction flow and conservative prompt rules, but
writes the current meeting-lab extraction schema.
"""
from __future__ import annotations
import argparse
import json
import re
import sys
import time
from pathlib import Path
from typing import Any
import requests
from src.meeting_lab.llm.prompts import build_extraction_prompt
DEFAULT_MODEL = "qwen3:8b"
DEFAULT_ENDPOINT = "http://localhost:11434/api/generate"
EXTRACTION_CATEGORIES = (
"facts",
"decisions",
"todos",
"questions",
"positions",
"technical",
)
NORMALIZED_CHUNK_RE = re.compile(r"^(chunk_\d+)_normalized\.txt$")
OUTPUT_SCHEMA = {
"chunk": {
"source_file": "string",
"summary": "Kurze, rein inhaltsbezogene Beschreibung des Chunks oder leer",
},
"participants": ["string"],
"topics": ["string"],
"facts": [
{
"speaker": "string oder null",
"statement": "Aussage möglichst nah am Wortlaut",
"status": "clear | unclear | contradictory",
"evidence": "Kurzes wörtliches oder nahezu wörtliches Textfragment",
}
],
"decisions": [
{
"decision": "string",
"evidence": "Kurzes Textfragment",
}
],
"todos": [
{
"task": "string",
"responsible": "string oder null",
"deadline": "string oder null",
"evidence": "Kurzes Textfragment",
}
],
"open_questions": [
{
"question": "string",
"evidence": "Kurzes Textfragment",
}
],
"technical_details": [
{
"subject": "string",
"statement": "Technische Aussage möglichst nah am Wortlaut",
"status": "clear | unclear | contradictory",
"evidence": "Kurzes Textfragment",
}
],
}
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Extract structured facts from normalized meeting chunks."
)
parser.add_argument(
"input",
type=Path,
help=(
"A chunk_XX_normalized.txt file or a directory containing "
"normalized chunk files"
),
)
parser.add_argument(
"-o",
"--output",
type=Path,
help=(
"Output JSON path for a single input file, or output directory "
"for a directory input"
),
)
parser.add_argument(
"--model",
default=DEFAULT_MODEL,
help=f"Ollama model name (default: {DEFAULT_MODEL})",
)
parser.add_argument(
"--endpoint",
default=DEFAULT_ENDPOINT,
help=f"Ollama generate endpoint (default: {DEFAULT_ENDPOINT})",
)
parser.add_argument(
"--timeout",
type=int,
default=1800,
help="HTTP timeout in seconds per chunk (default: 1800)",
)
parser.add_argument(
"--temperature",
type=float,
default=0.0,
help="Sampling temperature (default: 0.0)",
)
parser.add_argument(
"--num-predict",
type=int,
default=8192,
help="Maximum generated tokens per chunk (default: 8192)",
)
parser.add_argument(
"--num-ctx",
type=int,
default=32768,
help="Context window tokens per chunk (default: 32768)",
)
return parser.parse_args()
def build_prompt(source_name: str, transcript: str) -> str:
return build_extraction_prompt(source_name, transcript, OUTPUT_SCHEMA)
def call_ollama(
endpoint: str,
model: str,
prompt: str,
timeout: int,
temperature: float,
num_predict: int | None,
num_ctx: int | None,
) -> tuple[str, dict[str, Any]]:
options: dict[str, Any] = {
"temperature": temperature,
}
if num_predict is not None:
options["num_predict"] = num_predict
if num_ctx is not None:
options["num_ctx"] = num_ctx
payload = {
"model": model,
"prompt": prompt,
"think": False,
"stream": False,
"format": "json",
"options": options,
}
started = time.perf_counter()
response = requests.post(endpoint, json=payload, timeout=timeout)
elapsed = time.perf_counter() - started
response.raise_for_status()
data = response.json()
text = response_text_from_ollama_data(data)
if not isinstance(text, str) or not text.strip():
raise ValueError("Ollama returned no usable response text.")
metadata = {
"model": data.get("model", model),
"elapsed_seconds": round(elapsed, 3),
"total_duration_ns": data.get("total_duration"),
"load_duration_ns": data.get("load_duration"),
"prompt_eval_count": data.get("prompt_eval_count"),
"prompt_eval_duration_ns": data.get("prompt_eval_duration"),
"eval_count": data.get("eval_count"),
"eval_duration_ns": data.get("eval_duration"),
}
return text.strip(), metadata
def response_text_from_ollama_data(data: dict[str, Any]) -> str | None:
text = data.get("response")
if isinstance(text, str) and text.strip():
return text
message = data.get("message")
if isinstance(message, dict):
content = message.get("content")
if isinstance(content, str) and content.strip():
return content
thinking = data.get("thinking")
if isinstance(thinking, str) and thinking.strip():
return thinking
return text if isinstance(text, str) else None
def iter_json_object_candidates(text: str) -> list[str]:
candidates: list[str] = []
start: int | None = None
depth = 0
in_string = False
escaped = False
for index, char in enumerate(text):
if in_string:
if escaped:
escaped = False
elif char == "\\":
escaped = True
elif char == '"':
in_string = False
continue
if char == '"':
in_string = True
continue
if char == "{":
if depth == 0:
start = index
depth += 1
continue
if char == "}" and depth:
depth -= 1
if depth == 0 and start is not None:
candidates.append(text[start : index + 1])
start = None
return candidates
def parse_json_response(text: str) -> dict[str, Any]:
try:
parsed = json.loads(text)
except json.JSONDecodeError:
parsed = None
for candidate in iter_json_object_candidates(text):
try:
candidate_json = json.loads(candidate)
except json.JSONDecodeError:
continue
if isinstance(candidate_json, dict):
parsed = candidate_json
if parsed is None:
raise
if not isinstance(parsed, dict):
raise ValueError("The model response is valid JSON but not a JSON object.")
return parsed
def text_from_value(value: Any, preferred_keys: tuple[str, ...]) -> str:
if isinstance(value, str):
return value.strip()
if isinstance(value, dict):
parts: list[str] = []
for key in preferred_keys:
item = value.get(key)
if item is None:
continue
text = str(item).strip()
if text:
parts.append(text)
if parts:
return " | ".join(parts)
return json.dumps(value, ensure_ascii=False, sort_keys=True)
return str(value).strip()
def normalize_current_schema(result: dict[str, Any]) -> dict[str, list[str]]:
return {
"facts": [
text
for item in result.get("facts", [])
if (
text := text_from_value(
item,
("speaker", "statement", "status", "evidence"),
)
)
],
"decisions": [
text
for item in result.get("decisions", [])
if (
text := text_from_value(
item,
("decision", "evidence"),
)
)
],
"todos": [
text
for item in result.get("todos", [])
if (
text := text_from_value(
item,
("task", "responsible", "deadline", "evidence"),
)
)
],
"questions": [
text
for item in result.get("open_questions", result.get("questions", []))
if (
text := text_from_value(
item,
("question", "evidence"),
)
)
],
"positions": [
text
for item in result.get("positions", [])
if (
text := text_from_value(
item,
("speaker", "position", "statement", "evidence"),
)
)
],
"technical": [
text
for item in result.get("technical_details", result.get("technical", []))
if (
text := text_from_value(
item,
("subject", "statement", "status", "evidence"),
)
)
],
}
def extraction_path_for_chunk(chunk_path: Path, output_dir: Path | None = None) -> Path:
match = NORMALIZED_CHUNK_RE.match(chunk_path.name)
if not match:
raise ValueError(
f"Not a normalized chunk filename: {chunk_path.name}"
)
directory = output_dir or chunk_path.parent
return directory / f"{match.group(1)}_extraction.json"
def find_normalized_chunks(input_dir: Path) -> list[Path]:
return sorted(input_dir.glob("chunk_*_normalized.txt"))
def extract_chunk(
chunk_path: Path,
output_path: Path,
model: str,
endpoint: str,
timeout: int,
temperature: float,
num_predict: int | None,
num_ctx: int | None,
) -> dict[str, list[str]]:
transcript = chunk_path.read_text(encoding="utf-8-sig").strip()
if not transcript:
raise ValueError(f"The input file is empty: {chunk_path}")
prompt = build_prompt(chunk_path.name, transcript)
raw_text, _metadata = call_ollama(
endpoint=endpoint,
model=model,
prompt=prompt,
timeout=timeout,
temperature=temperature,
num_predict=num_predict,
num_ctx=num_ctx,
)
try:
parsed = parse_json_response(raw_text)
except (json.JSONDecodeError, ValueError) as exc:
raw_path = output_path.with_suffix(".raw.txt")
raw_path.write_text(raw_text + "\n", encoding="utf-8")
raise ValueError(
f"Model output was not valid JSON. Raw output saved to: {raw_path}"
) from exc
extraction = normalize_current_schema(parsed)
output_path.parent.mkdir(parents=True, exist_ok=True)
output_path.write_text(
json.dumps(extraction, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
return extraction
def extract_input(
input_path: Path,
output: Path | None,
model: str,
endpoint: str,
timeout: int,
temperature: float,
num_predict: int | None,
num_ctx: int | None,
) -> list[Path]:
if input_path.is_file():
output_path = output or extraction_path_for_chunk(input_path)
extract_chunk(
input_path,
output_path,
model,
endpoint,
timeout,
temperature,
num_predict,
num_ctx,
)
return [output_path]
if not input_path.is_dir():
raise FileNotFoundError(f"Input path not found: {input_path}")
chunk_paths = find_normalized_chunks(input_path)
if not chunk_paths:
raise ValueError(f"No normalized chunks found in {input_path}")
output_dir = output if output is not None else input_path
output_paths: list[Path] = []
for chunk_path in chunk_paths:
output_path = extraction_path_for_chunk(chunk_path, output_dir)
print(
f"Extracting {chunk_path.name} -> {output_path.name}",
flush=True,
)
extract_chunk(
chunk_path,
output_path,
model,
endpoint,
timeout,
temperature,
num_predict,
num_ctx,
)
output_paths.append(output_path)
return output_paths
def main() -> int:
args = parse_args()
try:
output_paths = extract_input(
input_path=args.input,
output=args.output,
model=args.model,
endpoint=args.endpoint,
timeout=args.timeout,
temperature=args.temperature,
num_predict=args.num_predict,
num_ctx=args.num_ctx,
)
except requests.ConnectionError:
print(
"Error: Ollama is not reachable. Is `ollama serve` running?",
file=sys.stderr,
)
return 1
except requests.Timeout:
print("Error: The Ollama request timed out.", file=sys.stderr)
return 1
except requests.HTTPError as exc:
print(f"Error: Ollama returned an HTTP error: {exc}", file=sys.stderr)
return 1
except (OSError, UnicodeError, ValueError) as exc:
print(f"Error: {exc}", file=sys.stderr)
return 1
print(f"Input: {args.input}")
print(f"Processed chunks: {len(output_paths)}")
print(f"Extraction JSON files: {len(output_paths)}")
for output_path in output_paths:
print(output_path)
return 0
if __name__ == "__main__":
raise SystemExit(main())
+51
View File
@@ -0,0 +1,51 @@
from __future__ import annotations
import json
from pathlib import Path
from typing import Any, Iterable
PROJECT_ROOT = Path(__file__).resolve().parents[3]
PROMPTS_DIR = PROJECT_ROOT / "prompts"
def load_prompt(name: str, prompts_dir: Path = PROMPTS_DIR) -> str:
path = prompts_dir / name
return path.read_text(encoding="utf-8").strip()
def load_existing_prompts(
names: Iterable[str],
prompts_dir: Path = PROMPTS_DIR,
) -> list[str]:
prompts: list[str] = []
for name in names:
text = load_prompt(name, prompts_dir)
if text:
prompts.append(text)
return prompts
def build_extraction_prompt(
source_name: str,
transcript: str,
output_schema: dict[str, Any],
task_prompt_names: Iterable[str] = ("decisions.md",),
prompts_dir: Path = PROMPTS_DIR,
) -> str:
schema_text = json.dumps(output_schema, ensure_ascii=False, indent=2)
prompt_parts = [
load_prompt("common.md", prompts_dir),
*load_existing_prompts(task_prompt_names, prompts_dir),
f"""Quelldatei:
{source_name}
Erwartete JSON-Struktur:
{schema_text}
TRANSKRIPT:
--- BEGINN TRANSKRIPT ---
{transcript}
--- ENDE TRANSKRIPT ---""",
]
return "\n\n".join(part for part in prompt_parts if part).strip() + "\n"
@@ -0,0 +1,251 @@
#!/usr/bin/env python3
"""
Conservatively normalize a meeting transcript without rewriting its meaning.
The normalizer only performs low-risk cleanup:
- removes isolated filler sounds such as "äh" and "ähm"
- collapses immediate duplicate words or short duplicate phrases
- normalizes whitespace
- records every changed block in a JSON change log
It deliberately does NOT remove modal words, qualifiers, negations,
dates, numbers, responsibilities, technical statements, or commitments.
Examples:
python normalize_transcript.py meeting_chunks/chunk_01.txt
python normalize_transcript.py meeting_chunks/chunk_01.txt \
--output meeting_chunks/chunk_01_normalized.txt \
--changes meeting_chunks/chunk_01_changes.json
"""
from __future__ import annotations
import argparse
import json
import re
import sys
from dataclasses import asdict, dataclass
from pathlib import Path
# Only clearly non-semantic filler sounds.
FILLER_PATTERN = re.compile(
r"(?i)(?<![\wäöüß])(?:ähm+|äh+|hm+|mhm+)(?![\wäöüß])"
)
# Immediate duplicate single word:
# "die die Anlage" -> "die Anlage"
DUPLICATE_WORD_PATTERN = re.compile(
r"(?i)\b([A-Za-zÄÖÜäöüß][\wÄÖÜäöüß'-]*)"
r"(?:[\s,;:]+)\1\b"
)
# Immediate duplicate short phrase of two to four words:
# "die Lüftung, die Lüftung" -> "die Lüftung"
DUPLICATE_PHRASE_PATTERN = re.compile(
r"(?i)\b("
r"[A-Za-zÄÖÜäöüß][\wÄÖÜäöüß'-]*"
r"(?:\s+[A-Za-zÄÖÜäöüß][\wÄÖÜäöüß'-]*){1,3}"
r")"
r"(?:\s*[,;:]\s*|\s+)\1\b"
)
MULTISPACE_PATTERN = re.compile(r"[ \t]{2,}")
SPACE_BEFORE_PUNCT_PATTERN = re.compile(r"\s+([,.;:!?])")
MULTI_BLANK_PATTERN = re.compile(r"\n{3,}")
@dataclass(frozen=True)
class Change:
block_number: int
original: str
normalized: str
removed_fillers: list[str]
duplicate_reductions: list[str]
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Conservatively normalize a meeting transcript."
)
parser.add_argument("input_file", type=Path, help="Transcript text file")
parser.add_argument(
"-o",
"--output",
type=Path,
help="Normalized output file; default: <input>_normalized.txt",
)
parser.add_argument(
"--changes",
type=Path,
help="Change log JSON; default: <input>_changes.json",
)
return parser.parse_args()
def split_blocks(text: str) -> list[str]:
"""
Preserve transcript structure by treating blank-line-separated paragraphs
as atomic blocks. If there are no blank lines, use individual non-empty
lines as blocks.
"""
normalized_newlines = text.replace("\r\n", "\n").replace("\r", "\n").strip()
paragraphs = [
part.strip()
for part in re.split(r"\n\s*\n+", normalized_newlines)
if part.strip()
]
if len(paragraphs) > 1:
return paragraphs
return [line.strip() for line in normalized_newlines.splitlines() if line.strip()]
def remove_fillers(text: str) -> tuple[str, list[str]]:
removed = [match.group(0) for match in FILLER_PATTERN.finditer(text)]
cleaned = FILLER_PATTERN.sub("", text)
return cleaned, removed
def reduce_duplicate_phrases(text: str) -> tuple[str, list[str]]:
reductions: list[str] = []
current = text
# Repeat until stable because one correction may expose another.
for _ in range(5):
changed = False
def phrase_replacer(match: re.Match[str]) -> str:
nonlocal changed
changed = True
reductions.append(match.group(0))
return match.group(1)
updated = DUPLICATE_PHRASE_PATTERN.sub(phrase_replacer, current)
current = updated
def word_replacer(match: re.Match[str]) -> str:
nonlocal changed
changed = True
reductions.append(match.group(0))
return match.group(1)
updated = DUPLICATE_WORD_PATTERN.sub(word_replacer, current)
current = updated
if not changed:
break
return current, reductions
def tidy_spacing(text: str) -> str:
text = MULTISPACE_PATTERN.sub(" ", text)
text = SPACE_BEFORE_PUNCT_PATTERN.sub(r"\1", text)
text = re.sub(r"([,;:])([^\s\n])", r"\1 \2", text)
text = re.sub(r"\(\s+", "(", text)
text = re.sub(r"\s+\)", ")", text)
return text.strip(" \t,;:")
def normalize_block(block: str, block_number: int) -> tuple[str, Change | None]:
original = block
result, removed_fillers = remove_fillers(original)
result, duplicate_reductions = reduce_duplicate_phrases(result)
result = tidy_spacing(result)
if result == original:
return result, None
return result, Change(
block_number=block_number,
original=original,
normalized=result,
removed_fillers=removed_fillers,
duplicate_reductions=duplicate_reductions,
)
def main() -> int:
args = parse_args()
try:
if not args.input_file.is_file():
raise FileNotFoundError(f"Input file not found: {args.input_file}")
source = args.input_file.read_text(encoding="utf-8-sig")
if not source.strip():
raise ValueError("The input file is empty.")
output_path = args.output or args.input_file.with_name(
f"{args.input_file.stem}_normalized.txt"
)
changes_path = args.changes or args.input_file.with_name(
f"{args.input_file.stem}_changes.json"
)
blocks = split_blocks(source)
normalized_blocks: list[str] = []
changes: list[Change] = []
for block_number, block in enumerate(blocks, start=1):
normalized, change = normalize_block(block, block_number)
normalized_blocks.append(normalized)
if change is not None:
changes.append(change)
normalized_text = "\n\n".join(normalized_blocks).strip() + "\n"
output_path.parent.mkdir(parents=True, exist_ok=True)
changes_path.parent.mkdir(parents=True, exist_ok=True)
output_path.write_text(normalized_text, encoding="utf-8")
report = {
"source_file": args.input_file.name,
"output_file": output_path.name,
"blocks_total": len(blocks),
"blocks_changed": len(changes),
"policy": {
"removed": [
"isolated filler sounds such as äh, ähm, hm, mhm",
"immediate duplicate words",
"immediate duplicate phrases of two to four words",
"redundant whitespace",
],
"explicitly_preserved": [
"negations",
"modal words and qualifiers",
"dates, times, numbers and quantities",
"technical statements",
"responsibilities",
"deadlines",
"decisions and commitments",
],
"principle": "When uncertain, leave the text unchanged.",
},
"changes": [asdict(change) for change in changes],
}
changes_path.write_text(
json.dumps(report, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
print(f"Source: {args.input_file}")
print(f"Normalized: {output_path}")
print(f"Change log: {changes_path}")
print(f"Blocks total: {len(blocks)}")
print(f"Blocks changed: {len(changes)}")
return 0
except (OSError, UnicodeError, ValueError) as exc:
print(f"Error: {exc}", file=sys.stderr)
return 1
if __name__ == "__main__":
raise SystemExit(main())
+223
View File
@@ -0,0 +1,223 @@
#!/usr/bin/env python3
"""
Build a readable Markdown protocol from chunk extraction JSON files.
"""
from __future__ import annotations
import argparse
import json
import re
import sys
from pathlib import Path
from typing import Any
EXTRACTION_CATEGORIES = (
"facts",
"decisions",
"todos",
"questions",
"positions",
"technical",
)
CATEGORY_TITLES = {
"facts": "Facts",
"decisions": "Decisions",
"todos": "Action Items",
"questions": "Open Questions",
"positions": "Positions",
"technical": "Technical",
}
CHUNK_NUMBER_RE = re.compile(r"chunk_(\d+)_")
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Build a Markdown meeting protocol from extraction JSON files."
)
parser.add_argument(
"input_dir",
type=Path,
help="Directory containing chunk_XX_extraction.json files",
)
parser.add_argument(
"-o",
"--output",
type=Path,
help="Output Markdown file; default: <input-dir>/meeting_protocol.md",
)
return parser.parse_args()
def chunk_sort_key(path: Path) -> tuple[int, str]:
match = CHUNK_NUMBER_RE.search(path.name)
if not match:
return (sys.maxsize, path.name)
return (int(match.group(1)), path.name)
def load_json(path: Path) -> Any:
return json.loads(path.read_text(encoding="utf-8-sig"))
def find_extraction_files(input_dir: Path) -> list[Path]:
return sorted(
input_dir.glob("chunk_*_extraction.json"),
key=chunk_sort_key,
)
def normalize_items(value: Any) -> list[str]:
if not isinstance(value, list):
return []
items: list[str] = []
for item in value:
if isinstance(item, str):
text = item.strip()
else:
text = json.dumps(item, ensure_ascii=False, sort_keys=True)
if text:
items.append(text)
return items
def collect_extractions(
extraction_files: list[Path],
) -> dict[str, list[str]]:
collected = {category: [] for category in EXTRACTION_CATEGORIES}
for path in extraction_files:
data = load_json(path)
if not isinstance(data, dict):
raise ValueError(
f"Extraction file must contain a JSON object: {path}"
)
for category in EXTRACTION_CATEGORIES:
collected[category].extend(
normalize_items(data.get(category, []))
)
return collected
def find_topic_files(input_dir: Path) -> list[Path]:
return sorted(
input_dir.glob("chunk_*_normalized_windowed_segments.json"),
key=chunk_sort_key,
)
def collect_topics(input_dir: Path) -> list[str]:
topics: list[str] = []
for path in find_topic_files(input_dir):
data = load_json(path)
if not isinstance(data, dict):
continue
source = data.get("source", {})
source_file = (
source.get("file")
if isinstance(source, dict)
else None
)
label = source_file or path.name
segments = data.get("segments", [])
if not isinstance(segments, list):
continue
for segment in segments:
if not isinstance(segment, dict):
continue
segment_id = segment.get("segment_id", "segment")
start_block = segment.get("start_block", "?")
end_block = segment.get("end_block", "?")
topics.append(
f"{label}: {segment_id} (blocks {start_block}-{end_block})"
)
return topics
def render_list(items: list[str]) -> list[str]:
if not items:
return ["- None recorded."]
return [f"- {item}" for item in items]
def build_protocol_markdown(
input_dir: Path,
extraction_files: list[Path],
) -> str:
extractions = collect_extractions(extraction_files)
topics = collect_topics(input_dir)
lines = [
"# Meeting Protocol",
"",
"## Topics",
*render_list(topics),
"",
"## Facts",
*render_list(extractions["facts"]),
"",
"## Decisions",
*render_list(extractions["decisions"]),
"",
"## Open Questions",
*render_list(extractions["questions"]),
"",
"## Action Items",
*render_list(extractions["todos"]),
"",
"## Positions",
*render_list(extractions["positions"]),
"",
"## Technical",
*render_list(extractions["technical"]),
"",
]
return "\n".join(lines)
def build_protocol(input_dir: Path, output_path: Path | None = None) -> Path:
if not input_dir.is_dir():
raise FileNotFoundError(f"Input directory not found: {input_dir}")
extraction_files = find_extraction_files(input_dir)
if not extraction_files:
raise ValueError(
f"No extraction JSON files found in {input_dir}"
)
output = output_path or input_dir / "meeting_protocol.md"
markdown = build_protocol_markdown(input_dir, extraction_files)
output.write_text(markdown, encoding="utf-8")
return output
def main() -> int:
args = parse_args()
try:
extraction_files = find_extraction_files(args.input_dir)
output_path = build_protocol(args.input_dir, args.output)
except (OSError, UnicodeError, ValueError, json.JSONDecodeError) as exc:
print(f"Error: {exc}", file=sys.stderr)
return 1
print(f"Input: {args.input_dir}")
print(f"Extraction JSON files: {len(extraction_files)}")
print(f"Protocol: {output_path}")
return 0
if __name__ == "__main__":
raise SystemExit(main())
+32
View File
@@ -0,0 +1,32 @@
# Decision Definition
A decision is any explicit agreement that creates a binding change in action,
process, responsibility, approval status, timing, or next step.
Included:
- substantive decisions
- organizational decisions
- process decisions
- approvals
- rejections
- deferrals
- explicit agreement not to decide yet
- explicit agreement to gather more information before deciding
Excluded:
- opinions
- preferences
- proposals without agreement
- open questions
- descriptions of the current state
- explanations without commitment
Important distinction:
- "No decision was reached" means the meeting ended without an agreed outcome.
- "The group decided to defer the decision" means the group explicitly agreed
on a process outcome: the substantive decision is postponed.
These are not equivalent.
@@ -0,0 +1,34 @@
# Prompt Engineering Methodology
Prompt engineering uses the Gold Standard corpus as the reference. The corpus is
not adjusted to make a prompt pass.
Rules:
1. Only one prompt change is allowed per iteration.
2. Only one gold test case may be optimized at a time.
3. Every prompt modification must be validated immediately.
4. A prompt modification is acceptable only if it improves the current target
and does not degrade any previously passing gold test.
5. Never modify `expected.json` to make a prompt pass.
6. Prompt engineering edits prompt files only. Python code changes require a
separate explicit task.
7. Maintain a prompt evolution log for every iteration.
8. If a prompt cannot improve a test after several small iterations, stop and
analyze the root cause.
9. Prompt changes must be generally applicable and must not special-case one
transcript.
10. If two consecutive prompt iterations fail to improve the current target,
stop further prompt modifications and classify the root cause.
11. If a prompt produces unexpected behavior, first verify whether the targeted
gold test has an objectively unique ground truth.
Prompt Version 2 baseline:
- `decision_simple`: passing
- `decision_deferred`: passing
- `decision_none`: passing
Prompt Version 2 adds explicit support for process decisions where the group
agrees to defer a substantive decision until additional information is
available.
+21
View File
@@ -0,0 +1,21 @@
# decision_deferred
Tests that a process decision to defer a substantive decision is still extracted
as a decision.
The transcript contains several candidate options for the weekly dashboard, but
the group does not choose any of them. Instead, Mira explicitly says not to
decide today and Jonas agrees that more input is needed.
Ground truth:
- No substantive dashboard schedule decision is reached.
- One process decision is reached: the substantive decision is deferred until
more information is available.
Typical LLM mistakes:
- Extracting Monday, Tuesday, or Friday as the chosen dashboard day.
- Treating "I like shorter" as an approval.
- Missing the deferral because the substantive decision is unresolved.
- Treating "no decision today" as equivalent to no decision at all.
@@ -0,0 +1,31 @@
{
"facts": [],
"decisions": [
{
"decision": "The substantive decision is deferred until more information is available.",
"evidence": "Mira: Okay, let's not decide this today. Jonas: Agreed, we need more input."
}
],
"todos": [
{
"task": "Lea will bring Dana's feedback about the weekly dashboard next time.",
"responsible": "Lea",
"deadline": "next time",
"evidence": "Lea: I will bring Dana's feedback next time."
}
],
"questions": [],
"positions": [
{
"speaker": "Jonas",
"position": "Jonas prefers moving the weekly dashboard to Monday morning.",
"evidence": "Jonas: I would prefer moving it to Monday morning."
},
{
"speaker": "Lea",
"position": "Lea thinks Monday is difficult for support.",
"evidence": "Lea: Monday is rough for support."
}
],
"technical": []
}
@@ -0,0 +1,19 @@
Mira: We need to talk about the weekly dashboard.
Jonas: I would prefer moving it to Monday morning.
Lea: Monday is rough for support. We usually have backlog cleanup then.
Mira: Tuesday might work, but I am not sure.
Jonas: Or we keep it on Friday and just make it shorter.
Lea: I like shorter, but I need to check with Dana first.
Mira: Okay, let's not decide this today.
Jonas: Agreed, we need more input.
Lea: I will bring Dana's feedback next time.
Mira: Thanks, that will help.
+21
View File
@@ -0,0 +1,21 @@
# decision_none
Tests a true decision-negative meeting segment.
The transcript contains discussion, competing preferences, and possible options
for the weekly dashboard. No participant approves an option, rejects an option
on behalf of the group, assigns a follow-up, agrees to gather more information,
or explicitly decides to defer the decision.
Ground truth:
- No decision was reached.
- No process decision was reached.
- No agreed next step was created.
Typical LLM mistakes:
- Treating a preference as a decision.
- Treating a proposed option as the selected option.
- Treating the topic change as an implicit deferral decision.
- Creating an action item for Dana even though she is only mentioned as absent.
+24
View File
@@ -0,0 +1,24 @@
{
"facts": [],
"decisions": [],
"todos": [],
"questions": [],
"positions": [
{
"speaker": "Jonas",
"position": "Jonas prefers moving the weekly dashboard to Monday morning.",
"evidence": "Jonas: I would prefer moving it to Monday morning."
},
{
"speaker": "Mira",
"position": "Mira thinks Tuesday might work better for support.",
"evidence": "Mira: Tuesday might work better for support."
},
{
"speaker": "Lea",
"position": "Lea suggests keeping Friday and making the dashboard shorter.",
"evidence": "Lea: Or we keep Friday and make the dashboard shorter."
}
],
"technical": []
}
+25
View File
@@ -0,0 +1,25 @@
Mira: We need to talk about the weekly dashboard.
Jonas: I would prefer moving it to Monday morning.
Lea: Monday is rough for support because backlog cleanup starts then.
Mira: Tuesday might work better for support.
Jonas: Tuesday is hard for sales, at least this month.
Lea: Or we keep Friday and make the dashboard shorter.
Mira: I am not convinced shorter solves the timing issue.
Jonas: I am not convinced Monday is actually a problem for everyone.
Lea: Dana might have a view, but she is not here.
Mira: We are circling now.
Jonas: Yes, I do not have anything else to add.
Lea: Same here.
Mira: Okay, let's move to the budget topic.
+11
View File
@@ -0,0 +1,11 @@
# decision_simple
Tests one explicit decision with clear agreement language.
The difficult part is separating the decision from nearby rationale about user confusion and from the non-decision statement that the copy can stay unchanged for now.
Typical LLM mistakes:
- Extracting the rationale as a separate decision.
- Treating "copy can stay as it is" as a formal decision.
- Losing the evidence that shows explicit agreement.
+13
View File
@@ -0,0 +1,13 @@
{
"facts": [],
"decisions": [
{
"decision": "The welcome email will be sent after account activation.",
"evidence": "So are we agreed that the welcome email moves to after activation? Ben: Agreed. Cara: Yes, let's do that."
}
],
"todos": [],
"questions": [],
"positions": [],
"technical": []
}
+19
View File
@@ -0,0 +1,19 @@
Anna: Before we leave the onboarding flow, can we settle the email step?
Ben: I still think the welcome email should go out after account activation, not before.
Cara: Yes, before activation it keeps confusing people.
Anna: So are we agreed that the welcome email moves to after activation?
Ben: Agreed.
Cara: Yes, let's do that.
Anna: Good. Then that is the decision for this release.
Ben: Separate note on the copy: I am not proposing any wording decision today.
Cara: Same here, no wording proposal from me.
Anna: Okay, then the only decision is the timing after activation.
+14
View File
@@ -0,0 +1,14 @@
# evil_meeting
Tests a deliberately difficult meeting with interruptions, corrections, topic switches, changed positions, absent referenced people, and near-decisions.
Every utterance is designed to trigger a common extraction failure. The meeting mentions Omar and Platform, but neither is a participant. It includes an explicit non-decision on migration and a real decision only on excluding FR-7 from Friday's batch.
Typical LLM mistakes:
- Extracting a migration decision even though the group says no migration decision today.
- Assigning Dana a todo even though she retracts it.
- Treating Omar as a participant or technical owner.
- Claiming Platform approved something despite being absent.
- Losing the correction from "old export" to "nightly CSV job" and from API export to CSV export.
- Treating Dana's opinion about rollout appearance as a fact or decision.
+73
View File
@@ -0,0 +1,73 @@
{
"facts": [
{
"fact": "Tenant FR-7 still used the nightly CSV job yesterday.",
"evidence": "Alex: Fine. So fact: tenant FR-7 still used the nightly CSV job yesterday."
},
{
"fact": "Omar is the customer contact, not the technical owner.",
"evidence": "Bea: Omar is the customer contact, not a participant here and not the technical owner."
},
{
"fact": "Platform is the technical owner, but nobody from Platform is in the meeting.",
"evidence": "Chen: The technical owner is still Platform, but nobody from Platform is in this call."
},
{
"fact": "Friday's rollout batch still includes DE-2 and NL-4.",
"evidence": "Alex: Good. Back to the portal rollout. Friday's batch still includes DE-2 and NL-4."
}
],
"decisions": [
{
"decision": "FR-7 is excluded from Friday's portal rollout batch.",
"evidence": "the portal rollout note will say FR-7 is excluded from Friday's batch. Bea: Agreed. Excluded from Friday's batch. Chen: Yes, put that in."
}
],
"todos": [
{
"task": "Check the FR-7 mapping table.",
"responsible": "Bea",
"deadline": "Thursday morning",
"evidence": "Bea: Yes, I will check it by Thursday morning."
}
],
"questions": [
{
"question": "Can FR-7 use the v2 mapping without a customer-side field rename?",
"evidence": "Alex: Open question: can FR-7 use the v2 mapping without a customer-side field rename?"
},
{
"question": "Can Omar confirm FR-7's preferred launch window?",
"evidence": "Dana: Also, can Omar confirm their preferred launch window?"
}
],
"positions": [
{
"speaker": "Dana",
"position": "Dana wants to migrate FR-7 but recognizes the mapping table may not be clean.",
"evidence": "Dana: I want to, but we do not know if the mapping table is clean."
},
{
"speaker": "Bea",
"position": "Bea changed her earlier position and now says not to migrate FR-7 until the mapping table is checked.",
"evidence": "I said last week we should migrate it. I am changing that. Do not migrate until the mapping table is checked."
},
{
"speaker": "Dana",
"position": "Dana thinks excluding FR-7 makes the rollout look messy.",
"evidence": "Dana: I personally think excluding FR-7 makes the rollout look messy."
}
],
"technical": [
{
"subject": "French export failure",
"statement": "The old nightly CSV job failed; the new exporter was not running on tenant FR-7.",
"evidence": "The old export failed. The new exporter was not running on that tenant."
},
{
"subject": "Export mapping tables",
"statement": "The API export uses the v2 mapping table, while the nightly CSV job uses the legacy table.",
"evidence": "the API export uses the v2 mapping table, the nightly CSV job uses the legacy table."
}
]
}
+61
View File
@@ -0,0 +1,61 @@
Alex: Okay, quick pass on the portal rollout. Wait, before that, the French CSV export broke again.
Bea: It did not break again. The old export failed. The new exporter was not running on that tenant.
Chen: Sorry, when you say old export, do you mean the nightly job?
Bea: Yes, the nightly CSV job. Not the API export.
Alex: Fine. So fact: tenant FR-7 still used the nightly CSV job yesterday.
Dana: I thought Omar owned that tenant.
Bea: Omar is the customer contact, not a participant here and not the technical owner.
Chen: The technical owner is still Platform, but nobody from Platform is in this call.
Alex: Should we decide to migrate FR-7 today?
Dana: I want to, but we do not know if the mapping table is clean.
Bea: Also, I said last week we should migrate it. I am changing that. Do not migrate until the mapping table is checked.
Chen: So no migration decision today?
Alex: Correct, no migration decision today.
Dana: But we can decide one thing: the portal rollout note will say FR-7 is excluded from Friday's batch.
Bea: Agreed. Excluded from Friday's batch.
Chen: Yes, put that in.
Alex: Action item: Bea checks the FR-7 mapping table by Thursday morning.
Bea: Yes, I will check it by Thursday morning.
Dana: And I will message Omar after Bea is done.
Alex: Hold on, after Bea is done is not a date.
Dana: Fair. Then no task for me yet. I need Bea's result first.
Chen: Technical note: the API export uses the v2 mapping table, the nightly CSV job uses the legacy table.
Bea: Correct.
Alex: Open question: can FR-7 use the v2 mapping without a customer-side field rename?
Dana: Also, can Omar confirm their preferred launch window?
Chen: Omar can answer that, but again he is not in this meeting.
Alex: Good. Back to the portal rollout. Friday's batch still includes DE-2 and NL-4.
Bea: Yes, those two are unchanged.
Dana: I personally think excluding FR-7 makes the rollout look messy.
Alex: Noted as Dana's view, not a decision.
Chen: And please don't write that Platform approved anything. They are absent.
+11
View File
@@ -0,0 +1,11 @@
# facts_simple
Tests extraction of objective facts from a short status update.
The transcript includes a question and a technical statement, but no decision or todo.
Typical LLM mistakes:
- Treating "Good" as approval of a decision.
- Turning "No decision needed today" into a decision.
- Missing that the scanner gateway details are technical as well as factual.
+32
View File
@@ -0,0 +1,32 @@
{
"facts": [
{
"fact": "The old scanner gateway is still running in aisle three.",
"evidence": "Tom: The old scanner gateway is still running in aisle three."
},
{
"fact": "Aisles one and two moved to the new gateway last week.",
"evidence": "Tom: Yes. Aisles one and two moved to the new gateway last week."
},
{
"fact": "The new gateway is handling live scans for receiving.",
"evidence": "Iris: The new gateway is already handling live scans for receiving."
}
],
"decisions": [],
"todos": [],
"questions": [
{
"question": "Is aisle three the only scanner gateway still left on the old gateway?",
"evidence": "Elena: Is that the only one left?"
}
],
"positions": [],
"technical": [
{
"subject": "Scanner gateway rollout",
"statement": "Aisle three remains on the old scanner gateway while aisles one and two use the new gateway.",
"evidence": "The old scanner gateway is still running in aisle three. Aisles one and two moved to the new gateway last week."
}
]
}
+19
View File
@@ -0,0 +1,19 @@
Elena: Quick status on the warehouse migration.
Tom: The old scanner gateway is still running in aisle three.
Elena: Is that the only one left?
Tom: Yes. Aisles one and two moved to the new gateway last week.
Iris: The new gateway is already handling live scans for receiving.
Elena: Good. Let's keep the rollout note factual.
Tom: No decision needed today.
Iris: Fine.
Elena: Anything else on warehouse?
Tom: No, that is all.
+11
View File
@@ -0,0 +1,11 @@
# facts_vs_positions
Tests separation of objective facts from personal opinions.
The transcript deliberately mixes numeric facts, named non-respondents, and subjective positions about rollout health.
Typical LLM mistakes:
- Treating Marta's opinion as an objective fact.
- Treating Leo's optimism as a fact.
- Extracting a rollout decision even though the group explicitly does not decide.
@@ -0,0 +1,42 @@
{
"facts": [
{
"fact": "The pilot survey closed yesterday with 42 responses.",
"evidence": "Hannah: The pilot survey closed yesterday with 42 responses."
},
{
"fact": "The average pilot survey rating was 3.8 out of 5.",
"evidence": "Leo: The average rating was 3.8 out of 5."
},
{
"fact": "Northwind, Verdan, and Eastport did not respond to the survey.",
"evidence": "Hannah: That part is true. Northwind, Verdan, and Eastport did not respond."
}
],
"decisions": [],
"todos": [],
"questions": [
{
"question": "Why does Marta think the survey result is weaker than it looks?",
"evidence": "Leo: Why?"
}
],
"positions": [
{
"speaker": "Marta",
"position": "Marta thinks the pilot survey result is weaker than it looks.",
"evidence": "Marta: I think that is weaker than it looks."
},
{
"speaker": "Leo",
"position": "Leo feels the pilot is healthy.",
"evidence": "Leo: I still feel the pilot is healthy."
},
{
"speaker": "Marta",
"position": "Marta thinks the rollout should slow down.",
"evidence": "Marta: I disagree. My view is that we should slow down the rollout."
}
],
"technical": []
}
@@ -0,0 +1,19 @@
Hannah: The pilot survey closed yesterday with 42 responses.
Leo: The average rating was 3.8 out of 5.
Marta: I think that is weaker than it looks.
Leo: Why?
Marta: Because three enterprise customers skipped the survey entirely.
Hannah: That part is true. Northwind, Verdan, and Eastport did not respond.
Leo: I still feel the pilot is healthy.
Marta: I disagree. My view is that we should slow down the rollout.
Hannah: Let's capture both views and not decide rollout speed today.
Leo: Okay, that matches my notes.
+11
View File
@@ -0,0 +1,11 @@
# mixed_small
Tests a compact realistic meeting containing all major extraction categories.
The transcript includes facts, one explicit decision, one action item, one open question, one opinion, and a technical constraint.
Typical LLM mistakes:
- Applying the demo-only decision to production.
- Treating Mateo's diagnosis as a fact instead of a position.
- Creating a long-term normalizer decision even though it is explicitly open.
+50
View File
@@ -0,0 +1,50 @@
{
"facts": [
{
"fact": "The staging import handled 12,000 rows last night.",
"evidence": "Priya: The staging import handled 12,000 rows last night."
},
{
"fact": "The staging import took 48 minutes.",
"evidence": "Mateo: It finished, but it took 48 minutes."
},
{
"fact": "The partner demo target is 30 minutes.",
"evidence": "Lena: Yes, that is still the demo target."
}
],
"decisions": [
{
"decision": "Address normalization will be disabled for the demo import only.",
"evidence": "Can we agree to disable address normalization for the demo import only? Lena: Yes, for the demo import only. Mateo: Agreed."
}
],
"todos": [
{
"task": "Update the demo import configuration.",
"responsible": "Mateo",
"deadline": "Friday noon",
"evidence": "Mateo: I will do that before Friday noon."
}
],
"questions": [
{
"question": "Whether a faster normalizer is needed after the demo.",
"evidence": "Lena: And the open question is whether we need a faster normalizer after the demo."
}
],
"positions": [
{
"speaker": "Mateo",
"position": "Mateo thinks address normalization is the slow part.",
"evidence": "Mateo: I think the slow part is address normalization."
}
],
"technical": [
{
"subject": "Demo import configuration",
"statement": "Address normalization is disabled only for the demo import; production imports keep full normalization.",
"evidence": "for the demo import only. Production imports keep the full normalization."
}
]
}
+23
View File
@@ -0,0 +1,23 @@
Priya: The staging import handled 12,000 rows last night.
Mateo: It finished, but it took 48 minutes.
Priya: The limit for the partner demo is 30 minutes, right?
Lena: Yes, that is still the demo target.
Mateo: I think the slow part is address normalization.
Priya: Can we agree to disable address normalization for the demo import only?
Lena: Yes, for the demo import only.
Mateo: Agreed. Production imports keep the full normalization.
Priya: Mateo, please update the demo config before Friday noon.
Mateo: I will do that before Friday noon.
Lena: And the open question is whether we need a faster normalizer after the demo.
Priya: Capture that, but no decision on the long-term fix today.
+11
View File
@@ -0,0 +1,11 @@
# question_simple
Tests extraction of an explicit open question.
The transcript also contains a non-task: Kai says he can ask finance but explicitly does not accept it as a task yet.
Typical LLM mistakes:
- Creating a todo for Kai despite his correction.
- Missing that the group decides to leave the issue open.
- Treating "support package" as enough information to answer the question.
+27
View File
@@ -0,0 +1,27 @@
{
"facts": [
{
"fact": "The vendor invoice came in this morning.",
"evidence": "Kai: The vendor invoice came in this morning."
},
{
"fact": "The invoice line item says support package.",
"evidence": "Kai: I don't know. The line item just says support package."
}
],
"decisions": [
{
"decision": "The support-hours invoice issue will remain an open question for now.",
"evidence": "Ruth: Fine. Let's leave it as an open question for now."
}
],
"todos": [],
"questions": [
{
"question": "Does the vendor invoice include the extra support hours from March?",
"evidence": "Ruth: Does it include the extra support hours from March?"
}
],
"positions": [],
"technical": []
}
+19
View File
@@ -0,0 +1,19 @@
Kai: The vendor invoice came in this morning.
Ruth: Does it include the extra support hours from March?
Kai: I don't know. The line item just says support package.
Ruth: Then that is still open.
Kai: I can ask finance, but I am not taking that as a task yet.
Ruth: Fine. Let's leave it as an open question for now.
Kai: Understood.
Ruth: Anything else on invoices?
Kai: No.
Ruth: Then next item.
+11
View File
@@ -0,0 +1,11 @@
# technical_simple
Tests technical extraction with a corrected diagnosis.
The transcript contrasts two possible causes: certificate expiry and runner configuration. The latter is confirmed as the cause.
Typical LLM mistakes:
- Reporting certificate expiry as the problem.
- Creating a fix decision even though the group explicitly says no fix is decided.
- Missing the distinction between fact and technical diagnosis.
+33
View File
@@ -0,0 +1,33 @@
{
"facts": [
{
"fact": "The mobile build failed on the staging runner.",
"evidence": "Sofia: The mobile build failed again on the staging runner."
},
{
"fact": "The certificate is valid until October.",
"evidence": "Nils: The certificate itself is valid until October."
}
],
"decisions": [],
"todos": [],
"questions": [
{
"question": "Is the mobile build failing with the same error as yesterday?",
"evidence": "Nils: Same error as yesterday?"
}
],
"positions": [],
"technical": [
{
"subject": "iOS staging build",
"statement": "The iOS job fails during code signing because the runner uses the old keychain path.",
"evidence": "The iOS job now fails during code signing. The problem is that the runner uses the old keychain path."
},
{
"subject": "Failure classification",
"statement": "The failure is a runner configuration issue, not a certificate expiry issue.",
"evidence": "So it is a runner configuration issue, not a certificate expiry issue. Sofia: Exactly."
}
]
}
@@ -0,0 +1,19 @@
Sofia: The mobile build failed again on the staging runner.
Nils: Same error as yesterday?
Sofia: No, different. The iOS job now fails during code signing.
Nils: The certificate itself is valid until October.
Sofia: Right. The problem is that the runner uses the old keychain path.
Nils: So it is a runner configuration issue, not a certificate expiry issue.
Sofia: Exactly.
Nils: We are not deciding the fix today.
Sofia: Okay.
Nils: Next item.
+11
View File
@@ -0,0 +1,11 @@
# todo_negative
Tests that vague "someone should" language is not an action item.
The transcript contains a real decision to keep the issue on the risk list, but no assigned task.
Typical LLM mistakes:
- Creating a todo with "someone" as owner.
- Assigning Paula, Ravi, or Sam even though they explicitly do not take ownership.
- Ignoring the explicit "no owner for now" correction.
+22
View File
@@ -0,0 +1,22 @@
{
"facts": [
{
"fact": "The customer export takes longer than Paula expected.",
"evidence": "Paula: The customer export takes longer than I expected."
},
{
"fact": "There is no owner for the customer export issue for now.",
"evidence": "Paula: Right, no owner for now."
}
],
"decisions": [
{
"decision": "The customer export issue will stay on the risk list for now.",
"evidence": "Sam: Then let's just keep it on the risk list. Paula: Right, no owner for now."
}
],
"todos": [],
"questions": [],
"positions": [],
"technical": []
}
+19
View File
@@ -0,0 +1,19 @@
Paula: The customer export takes longer than I expected.
Ravi: Someone should probably look at it.
Sam: Yes, maybe after the release freeze.
Paula: I don't have capacity this week.
Ravi: Same here.
Sam: Then let's just keep it on the risk list.
Paula: Right, no owner for now.
Ravi: We can revisit it in planning.
Sam: Okay.
Paula: Next item.
+11
View File
@@ -0,0 +1,11 @@
# todo_simple
Tests one clear action item with responsible person and deadline.
The difficult part is not adding extra scope: Nina explicitly says she will not touch the layout.
Typical LLM mistakes:
- Adding a layout update as a task.
- Dropping the deadline.
- Turning Omar's request into the task evidence instead of Nina's commitment.
+20
View File
@@ -0,0 +1,20 @@
{
"facts": [
{
"fact": "The beta signup page points to the old privacy note.",
"evidence": "Omar: The beta signup page still points to the old privacy note."
}
],
"decisions": [],
"todos": [
{
"task": "Update the privacy link on the beta signup page.",
"responsible": "Nina",
"deadline": "Thursday noon",
"evidence": "Nina: Yes, I will update the privacy link by Thursday noon."
}
],
"questions": [],
"positions": [],
"technical": []
}
+19
View File
@@ -0,0 +1,19 @@
Omar: The beta signup page still points to the old privacy note.
Nina: Yes, I saw that yesterday.
Omar: Can you update the link before the partner demo?
Nina: Yes, I will update the privacy link by Thursday noon.
Omar: Great. Nothing else on that page from my side.
Nina: I will only touch the link, not the layout.
Omar: Fine.
Nina: Then I have what I need.
Omar: We can move on.
Nina: Yes.
+50
View File
@@ -0,0 +1,50 @@
import json
import tempfile
import unittest
from pathlib import Path
from src.meeting_lab.chunking.chunk_transcript import (
build_chunks,
read_transcript_blocks,
rendered_length,
)
class ChunkTranscriptTests(unittest.TestCase):
def test_whisper_json_uses_segments_not_aggregate_text(self) -> None:
data = {
"text": "alpha beta gamma",
"segments": [
{"text": "alpha"},
{"text": "beta"},
{"text": "gamma"},
],
}
with tempfile.TemporaryDirectory() as directory:
input_file = Path(directory) / "meeting.json"
input_file.write_text(json.dumps(data), encoding="utf-8")
blocks = read_transcript_blocks(input_file)
self.assertEqual(blocks, ["alpha", "beta", "gamma"])
self.assertNotIn("alpha beta gamma", blocks)
def test_chunks_do_not_share_later_blocks_without_overlap(self) -> None:
blocks = [f"block-{number:02d}-" + ("x" * 20) for number in range(10)]
chunks = build_chunks(
blocks=blocks,
target_chars=60,
max_chars=80,
min_chars=30,
overlap_blocks=0,
)
flattened = [block for chunk in chunks for block in chunk]
self.assertEqual(flattened, blocks)
self.assertEqual(len(flattened), len(set(flattened)))
self.assertTrue(all(rendered_length(chunk) <= 80 for chunk in chunks))
if __name__ == "__main__":
unittest.main()
+133
View File
@@ -0,0 +1,133 @@
import json
import tempfile
import unittest
from pathlib import Path
from src.meeting_lab.extraction.extract_chunks import (
EXTRACTION_CATEGORIES,
build_prompt,
extraction_path_for_chunk,
normalize_current_schema,
parse_json_response,
)
from src.meeting_lab.protocol.build_protocol import build_protocol
class ExtractionProtocolTests(unittest.TestCase):
def test_extraction_path_drops_normalized_suffix(self) -> None:
path = Path("chunks/chunk_01_normalized.txt")
self.assertEqual(
extraction_path_for_chunk(path),
Path("chunks/chunk_01_extraction.json"),
)
def test_normalize_current_schema_maps_legacy_response(self) -> None:
result = normalize_current_schema(
{
"facts": [
{
"speaker": "A",
"statement": "Fact one",
"status": "clear",
"evidence": "Fact",
}
],
"decisions": [
{
"decision": "Decision one",
"evidence": "Decision",
}
],
"todos": [
{
"task": "Todo one",
"responsible": "B",
"deadline": None,
"evidence": "Todo",
}
],
"open_questions": [
{
"question": "Question one",
"evidence": "Question",
}
],
"technical_details": [
{
"subject": "System",
"statement": "Technical one",
"status": "clear",
"evidence": "Technical",
}
],
}
)
self.assertEqual(set(result), set(EXTRACTION_CATEGORIES))
self.assertEqual(result["positions"], [])
self.assertIn("Fact one", result["facts"][0])
self.assertIn("Decision one", result["decisions"][0])
self.assertIn("Todo one", result["todos"][0])
self.assertIn("Question one", result["questions"][0])
self.assertIn("Technical one", result["technical"][0])
def test_parse_json_response_uses_final_object_after_thinking(self) -> None:
text = """
Thinking:
I will reason about the transcript first.
{"facts": ["draft"], "decisions": []}
Final answer:
{
"facts": ["final"],
"decisions": [],
"todos": [],
"questions": [],
"positions": [],
"technical": []
}
"""
parsed = parse_json_response(text)
self.assertEqual(parsed["facts"], ["final"])
self.assertEqual(set(parsed), set(EXTRACTION_CATEGORIES))
def test_build_prompt_includes_common_and_decision_prompt_files(self) -> None:
prompt = build_prompt("transcript.txt", "Anna: Agreed.")
self.assertIn("Du extrahierst Informationen aus Meeting-Transkripten.", prompt)
self.assertIn("You extract decisions from meeting transcript text.", prompt)
self.assertIn("Extract each decision as one atomic commitment.", prompt)
self.assertIn("Anna: Agreed.", prompt)
def test_build_protocol_groups_extraction_items(self) -> None:
with tempfile.TemporaryDirectory() as directory:
input_dir = Path(directory)
(input_dir / "chunk_01_extraction.json").write_text(
json.dumps(
{
"facts": ["Fact one"],
"decisions": ["Decision one"],
"todos": ["Todo one"],
"questions": ["Question one"],
"positions": [],
"technical": [],
}
),
encoding="utf-8",
)
output_path = build_protocol(input_dir)
markdown = output_path.read_text(encoding="utf-8")
self.assertIn("# Meeting Protocol", markdown)
self.assertIn("## Facts\n- Fact one", markdown)
self.assertIn("## Decisions\n- Decision one", markdown)
self.assertIn("## Open Questions\n- Question one", markdown)
self.assertIn("## Action Items\n- Todo one", markdown)
if __name__ == "__main__":
unittest.main()
+51
View File
@@ -0,0 +1,51 @@
import json
import tempfile
import unittest
from pathlib import Path
from scripts.run_gold_test import read_json_object, scenario_paths, validate_required_keys
class GoldRunnerTests(unittest.TestCase):
def test_missing_transcript_is_reported(self) -> None:
with tempfile.TemporaryDirectory() as directory:
scenario = Path(directory)
(scenario / "expected.json").write_text("{}", encoding="utf-8")
with self.assertRaisesRegex(FileNotFoundError, "Missing transcript.txt"):
scenario_paths(scenario)
def test_missing_expected_is_reported(self) -> None:
with tempfile.TemporaryDirectory() as directory:
scenario = Path(directory)
(scenario / "transcript.txt").write_text("A: Hello.", encoding="utf-8")
with self.assertRaisesRegex(FileNotFoundError, "Missing expected.json"):
scenario_paths(scenario)
def test_invalid_expected_json_is_reported(self) -> None:
with tempfile.TemporaryDirectory() as directory:
path = Path(directory) / "expected.json"
path.write_text("{invalid", encoding="utf-8")
with self.assertRaisesRegex(ValueError, "Invalid JSON"):
read_json_object(path)
def test_missing_required_schema_keys_are_reported(self) -> None:
with tempfile.TemporaryDirectory() as directory:
path = Path(directory) / "expected.json"
data = {
"facts": [],
"decisions": [],
"todos": [],
"questions": [],
"positions": [],
}
path.write_text(json.dumps(data), encoding="utf-8")
with self.assertRaisesRegex(ValueError, "technical"):
validate_required_keys(read_json_object(path), path)
if __name__ == "__main__":
unittest.main()