diff --git a/.gitignore b/.gitignore
index aaffe1d..a516c1d 100644
--- a/.gitignore
+++ b/.gitignore
@@ -32,6 +32,13 @@ htmlcov/
experiments/**/output/
experiments/**/results/
+# Pipeline runtime artifacts
+samples/raw/
+samples/whisper/
+**/chunk_*_extraction.json
+**/meeting_protocol.md
+**/*.raw.txt
+
# Lokale Meetings (niemals versionieren)
meeting_data/
recordings/
diff --git a/scripts/clean_whisper_json.py b/scripts/clean_whisper_json.py
new file mode 100644
index 0000000..a96f861
--- /dev/null
+++ b/scripts/clean_whisper_json.py
@@ -0,0 +1,139 @@
+import argparse
+import json
+import math
+from pathlib import Path
+from typing import Any
+
+
+def is_nonfinite_number(value: Any) -> bool:
+ """Prüft auf NaN sowie positive oder negative Unendlichkeit."""
+ return isinstance(value, float) and not math.isfinite(value)
+
+
+def sanitize_nonfinite_values(value: Any) -> Any:
+ """
+ Ersetzt NaN und Infinity rekursiv durch None.
+
+ None wird in JSON als null geschrieben.
+ """
+ if is_nonfinite_number(value):
+ return None
+
+ if isinstance(value, dict):
+ return {
+ key: sanitize_nonfinite_values(item)
+ for key, item in value.items()
+ }
+
+ if isinstance(value, list):
+ return [
+ sanitize_nonfinite_values(item)
+ for item in value
+ ]
+
+ return value
+
+
+def is_invalid_empty_segment(segment: dict[str, Any]) -> bool:
+ """Erkennt technisch leere Whisper-Artefakte."""
+
+ text = str(segment.get("text", "")).strip()
+ start = segment.get("start")
+ end = segment.get("end")
+ avg_logprob = segment.get("avg_logprob")
+
+ logprob_is_nan = (
+ isinstance(avg_logprob, float)
+ and math.isnan(avg_logprob)
+ )
+
+ return (
+ not text
+ and start == end
+ and logprob_is_nan
+ )
+
+
+def clean_whisper_json(
+ input_path: Path,
+ output_path: Path,
+) -> None:
+ with input_path.open("r", encoding="utf-8") as file:
+ data = json.load(file)
+
+ original_segments = data.get("segments", [])
+
+ cleaned_segments = [
+ segment
+ for segment in original_segments
+ if not is_invalid_empty_segment(segment)
+ ]
+
+ for new_id, segment in enumerate(cleaned_segments):
+ segment["id"] = new_id
+
+ data["segments"] = cleaned_segments
+
+ data["text"] = " ".join(
+ str(segment.get("text", "")).strip()
+ for segment in cleaned_segments
+ if str(segment.get("text", "")).strip()
+ )
+
+ # Alle noch vorhandenen NaN-/Infinity-Werte durch null ersetzen.
+ data = sanitize_nonfinite_values(data)
+
+ # Erst vollständig in einen String serialisieren.
+ # Dadurch bleibt keine unvollständige Ausgabedatei zurück,
+ # falls doch noch ein Fehler auftritt.
+ json_content = json.dumps(
+ data,
+ ensure_ascii=False,
+ indent=2,
+ allow_nan=False,
+ )
+
+ output_path.write_text(
+ json_content + "\n",
+ encoding="utf-8",
+ )
+
+ removed = len(original_segments) - len(cleaned_segments)
+
+ print(f"Eingabedatei: {input_path}")
+ print(f"Ausgabedatei: {output_path}")
+ print(f"Segmente vorher: {len(original_segments)}")
+ print(f"Segmente nachher: {len(cleaned_segments)}")
+ print(f"Segmente entfernt: {removed}")
+
+
+def main() -> None:
+ parser = argparse.ArgumentParser(
+ description=(
+ "Entfernt technisch leere Artefakte aus "
+ "einer MLX-Whisper-JSON-Datei."
+ )
+ )
+ parser.add_argument(
+ "input",
+ type=Path,
+ help="Whisper-JSON-Datei",
+ )
+ parser.add_argument(
+ "-o",
+ "--output",
+ type=Path,
+ help="Ausgabedatei",
+ )
+
+ args = parser.parse_args()
+
+ output_path = args.output or args.input.with_name(
+ f"{args.input.stem}_cleaned.json"
+ )
+
+ clean_whisper_json(args.input, output_path)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/src/meeting_lab/extraction/extract_chunks.py b/src/meeting_lab/extraction/extract_chunks.py
new file mode 100644
index 0000000..f51f6f0
--- /dev/null
+++ b/src/meeting_lab/extraction/extract_chunks.py
@@ -0,0 +1,550 @@
+#!/usr/bin/env python3
+"""
+Extract structured meeting information from normalized transcript chunks.
+
+This module is adapted from /Users/tazl/quick-whisper-test/summarize.py.
+It keeps the same Ollama extraction flow and conservative prompt rules, but
+writes the current meeting-lab extraction schema.
+"""
+
+from __future__ import annotations
+
+import argparse
+import json
+import re
+import sys
+import time
+from pathlib import Path
+from typing import Any
+
+import requests
+
+
+DEFAULT_MODEL = "qwen3:8b"
+DEFAULT_ENDPOINT = "http://localhost:11434/api/generate"
+
+EXTRACTION_CATEGORIES = (
+ "facts",
+ "decisions",
+ "todos",
+ "questions",
+ "positions",
+ "technical",
+)
+
+NORMALIZED_CHUNK_RE = re.compile(r"^(chunk_\d+)_normalized\.txt$")
+
+
+SYSTEM_INSTRUCTION = """
+Du extrahierst Informationen aus Meeting-Transkripten.
+
+Arbeite ausschließlich mit dem vorgelegten Transkript.
+Verwende kein eigenes Fachwissen, keine Vermutungen und keine üblichen
+Funktionsweisen technischer Systeme.
+
+Regeln:
+1. Erfinde nichts.
+2. Interpretiere technische Aussagen nicht über den Wortlaut hinaus.
+3. Korrigiere keine Aussagen anhand vermeintlichen Weltwissens.
+4. Wenn etwas widersprüchlich oder unklar ist, kennzeichne es als unklar.
+5. Übernimm wichtige technische Aussagen möglichst nah am Wortlaut.
+6. Nenne bei Fakten nach Möglichkeit den Sprecher.
+7. Ein Beschluss ist nur dann ein Beschluss, wenn im Text eine Einigung,
+ Freigabe oder verbindliche Festlegung erkennbar ist.
+8. Eine Aufgabe ist nur dann eine Aufgabe, wenn eine Handlung und möglichst
+ eine verantwortliche Person oder Organisation erkennbar sind.
+9. Gib ausschließlich gültiges JSON aus. Kein Markdown, keine Erläuterungen.
+""".strip()
+
+
+OUTPUT_SCHEMA = {
+ "chunk": {
+ "source_file": "string",
+ "summary": "Kurze, rein inhaltsbezogene Beschreibung des Chunks oder leer",
+ },
+ "participants": ["string"],
+ "topics": ["string"],
+ "facts": [
+ {
+ "speaker": "string oder null",
+ "statement": "Aussage möglichst nah am Wortlaut",
+ "status": "clear | unclear | contradictory",
+ "evidence": "Kurzes wörtliches oder nahezu wörtliches Textfragment",
+ }
+ ],
+ "decisions": [
+ {
+ "decision": "string",
+ "evidence": "Kurzes Textfragment",
+ }
+ ],
+ "todos": [
+ {
+ "task": "string",
+ "responsible": "string oder null",
+ "deadline": "string oder null",
+ "evidence": "Kurzes Textfragment",
+ }
+ ],
+ "open_questions": [
+ {
+ "question": "string",
+ "evidence": "Kurzes Textfragment",
+ }
+ ],
+ "technical_details": [
+ {
+ "subject": "string",
+ "statement": "Technische Aussage möglichst nah am Wortlaut",
+ "status": "clear | unclear | contradictory",
+ "evidence": "Kurzes Textfragment",
+ }
+ ],
+}
+
+
+def parse_args() -> argparse.Namespace:
+ parser = argparse.ArgumentParser(
+ description="Extract structured facts from normalized meeting chunks."
+ )
+ parser.add_argument(
+ "input",
+ type=Path,
+ help=(
+ "A chunk_XX_normalized.txt file or a directory containing "
+ "normalized chunk files"
+ ),
+ )
+ parser.add_argument(
+ "-o",
+ "--output",
+ type=Path,
+ help=(
+ "Output JSON path for a single input file, or output directory "
+ "for a directory input"
+ ),
+ )
+ parser.add_argument(
+ "--model",
+ default=DEFAULT_MODEL,
+ help=f"Ollama model name (default: {DEFAULT_MODEL})",
+ )
+ parser.add_argument(
+ "--endpoint",
+ default=DEFAULT_ENDPOINT,
+ help=f"Ollama generate endpoint (default: {DEFAULT_ENDPOINT})",
+ )
+ parser.add_argument(
+ "--timeout",
+ type=int,
+ default=1800,
+ help="HTTP timeout in seconds per chunk (default: 1800)",
+ )
+ parser.add_argument(
+ "--temperature",
+ type=float,
+ default=0.0,
+ help="Sampling temperature (default: 0.0)",
+ )
+ parser.add_argument(
+ "--num-predict",
+ type=int,
+ default=8192,
+ help="Maximum generated tokens per chunk (default: 8192)",
+ )
+ parser.add_argument(
+ "--num-ctx",
+ type=int,
+ default=32768,
+ help="Context window tokens per chunk (default: 32768)",
+ )
+ return parser.parse_args()
+
+
+def build_prompt(source_name: str, transcript: str) -> str:
+ schema_text = json.dumps(OUTPUT_SCHEMA, ensure_ascii=False, indent=2)
+
+ return f"""{SYSTEM_INSTRUCTION}
+
+Quelldatei:
+{source_name}
+
+Erwartete JSON-Struktur:
+{schema_text}
+
+Hinweise zur Ausgabe:
+- Alle obersten Schlüssel müssen vorhanden sein.
+- Verwende leere Listen, wenn keine Einträge vorhanden sind.
+- Verwende null, wenn Verantwortliche, Sprecher oder Termine nicht erkennbar sind.
+- "evidence" muss sich eng am Transkript orientieren.
+- Ersetze technische Aussagen niemals durch eine vermeintlich korrektere Erklärung.
+- Confidence-Werte sind ausdrücklich nicht erwünscht.
+
+TRANSKRIPT:
+--- BEGINN TRANSKRIPT ---
+{transcript}
+--- ENDE TRANSKRIPT ---
+"""
+
+
+def call_ollama(
+ endpoint: str,
+ model: str,
+ prompt: str,
+ timeout: int,
+ temperature: float,
+ num_predict: int | None,
+ num_ctx: int | None,
+) -> tuple[str, dict[str, Any]]:
+ options: dict[str, Any] = {
+ "temperature": temperature,
+ }
+ if num_predict is not None:
+ options["num_predict"] = num_predict
+ if num_ctx is not None:
+ options["num_ctx"] = num_ctx
+
+ payload = {
+ "model": model,
+ "prompt": prompt,
+ "think": False,
+ "stream": False,
+ "format": "json",
+ "options": options,
+ }
+
+ started = time.perf_counter()
+ response = requests.post(endpoint, json=payload, timeout=timeout)
+ elapsed = time.perf_counter() - started
+ response.raise_for_status()
+
+ data = response.json()
+ text = response_text_from_ollama_data(data)
+ if not isinstance(text, str) or not text.strip():
+ raise ValueError("Ollama returned no usable response text.")
+
+ metadata = {
+ "model": data.get("model", model),
+ "elapsed_seconds": round(elapsed, 3),
+ "total_duration_ns": data.get("total_duration"),
+ "load_duration_ns": data.get("load_duration"),
+ "prompt_eval_count": data.get("prompt_eval_count"),
+ "prompt_eval_duration_ns": data.get("prompt_eval_duration"),
+ "eval_count": data.get("eval_count"),
+ "eval_duration_ns": data.get("eval_duration"),
+ }
+ return text.strip(), metadata
+
+
+def response_text_from_ollama_data(data: dict[str, Any]) -> str | None:
+ text = data.get("response")
+ if isinstance(text, str) and text.strip():
+ return text
+
+ message = data.get("message")
+ if isinstance(message, dict):
+ content = message.get("content")
+ if isinstance(content, str) and content.strip():
+ return content
+
+ thinking = data.get("thinking")
+ if isinstance(thinking, str) and thinking.strip():
+ return thinking
+
+ return text if isinstance(text, str) else None
+
+
+def iter_json_object_candidates(text: str) -> list[str]:
+ candidates: list[str] = []
+ start: int | None = None
+ depth = 0
+ in_string = False
+ escaped = False
+
+ for index, char in enumerate(text):
+ if in_string:
+ if escaped:
+ escaped = False
+ elif char == "\\":
+ escaped = True
+ elif char == '"':
+ in_string = False
+ continue
+
+ if char == '"':
+ in_string = True
+ continue
+
+ if char == "{":
+ if depth == 0:
+ start = index
+ depth += 1
+ continue
+
+ if char == "}" and depth:
+ depth -= 1
+ if depth == 0 and start is not None:
+ candidates.append(text[start : index + 1])
+ start = None
+
+ return candidates
+
+
+def parse_json_response(text: str) -> dict[str, Any]:
+ try:
+ parsed = json.loads(text)
+ except json.JSONDecodeError:
+ parsed = None
+ for candidate in iter_json_object_candidates(text):
+ try:
+ candidate_json = json.loads(candidate)
+ except json.JSONDecodeError:
+ continue
+ if isinstance(candidate_json, dict):
+ parsed = candidate_json
+ if parsed is None:
+ raise
+
+ if not isinstance(parsed, dict):
+ raise ValueError("The model response is valid JSON but not a JSON object.")
+ return parsed
+
+
+def text_from_value(value: Any, preferred_keys: tuple[str, ...]) -> str:
+ if isinstance(value, str):
+ return value.strip()
+
+ if isinstance(value, dict):
+ parts: list[str] = []
+ for key in preferred_keys:
+ item = value.get(key)
+ if item is None:
+ continue
+ text = str(item).strip()
+ if text:
+ parts.append(text)
+ if parts:
+ return " | ".join(parts)
+ return json.dumps(value, ensure_ascii=False, sort_keys=True)
+
+ return str(value).strip()
+
+
+def normalize_current_schema(result: dict[str, Any]) -> dict[str, list[str]]:
+ return {
+ "facts": [
+ text
+ for item in result.get("facts", [])
+ if (
+ text := text_from_value(
+ item,
+ ("speaker", "statement", "status", "evidence"),
+ )
+ )
+ ],
+ "decisions": [
+ text
+ for item in result.get("decisions", [])
+ if (
+ text := text_from_value(
+ item,
+ ("decision", "evidence"),
+ )
+ )
+ ],
+ "todos": [
+ text
+ for item in result.get("todos", [])
+ if (
+ text := text_from_value(
+ item,
+ ("task", "responsible", "deadline", "evidence"),
+ )
+ )
+ ],
+ "questions": [
+ text
+ for item in result.get("open_questions", result.get("questions", []))
+ if (
+ text := text_from_value(
+ item,
+ ("question", "evidence"),
+ )
+ )
+ ],
+ "positions": [
+ text
+ for item in result.get("positions", [])
+ if (
+ text := text_from_value(
+ item,
+ ("speaker", "position", "statement", "evidence"),
+ )
+ )
+ ],
+ "technical": [
+ text
+ for item in result.get("technical_details", result.get("technical", []))
+ if (
+ text := text_from_value(
+ item,
+ ("subject", "statement", "status", "evidence"),
+ )
+ )
+ ],
+ }
+
+
+def extraction_path_for_chunk(chunk_path: Path, output_dir: Path | None = None) -> Path:
+ match = NORMALIZED_CHUNK_RE.match(chunk_path.name)
+ if not match:
+ raise ValueError(
+ f"Not a normalized chunk filename: {chunk_path.name}"
+ )
+ directory = output_dir or chunk_path.parent
+ return directory / f"{match.group(1)}_extraction.json"
+
+
+def find_normalized_chunks(input_dir: Path) -> list[Path]:
+ return sorted(input_dir.glob("chunk_*_normalized.txt"))
+
+
+def extract_chunk(
+ chunk_path: Path,
+ output_path: Path,
+ model: str,
+ endpoint: str,
+ timeout: int,
+ temperature: float,
+ num_predict: int | None,
+ num_ctx: int | None,
+) -> dict[str, list[str]]:
+ transcript = chunk_path.read_text(encoding="utf-8-sig").strip()
+ if not transcript:
+ raise ValueError(f"The input file is empty: {chunk_path}")
+
+ prompt = build_prompt(chunk_path.name, transcript)
+ raw_text, _metadata = call_ollama(
+ endpoint=endpoint,
+ model=model,
+ prompt=prompt,
+ timeout=timeout,
+ temperature=temperature,
+ num_predict=num_predict,
+ num_ctx=num_ctx,
+ )
+
+ try:
+ parsed = parse_json_response(raw_text)
+ except (json.JSONDecodeError, ValueError) as exc:
+ raw_path = output_path.with_suffix(".raw.txt")
+ raw_path.write_text(raw_text + "\n", encoding="utf-8")
+ raise ValueError(
+ f"Model output was not valid JSON. Raw output saved to: {raw_path}"
+ ) from exc
+
+ extraction = normalize_current_schema(parsed)
+ output_path.parent.mkdir(parents=True, exist_ok=True)
+ output_path.write_text(
+ json.dumps(extraction, ensure_ascii=False, indent=2) + "\n",
+ encoding="utf-8",
+ )
+ return extraction
+
+
+def extract_input(
+ input_path: Path,
+ output: Path | None,
+ model: str,
+ endpoint: str,
+ timeout: int,
+ temperature: float,
+ num_predict: int | None,
+ num_ctx: int | None,
+) -> list[Path]:
+ if input_path.is_file():
+ output_path = output or extraction_path_for_chunk(input_path)
+ extract_chunk(
+ input_path,
+ output_path,
+ model,
+ endpoint,
+ timeout,
+ temperature,
+ num_predict,
+ num_ctx,
+ )
+ return [output_path]
+
+ if not input_path.is_dir():
+ raise FileNotFoundError(f"Input path not found: {input_path}")
+
+ chunk_paths = find_normalized_chunks(input_path)
+ if not chunk_paths:
+ raise ValueError(f"No normalized chunks found in {input_path}")
+
+ output_dir = output if output is not None else input_path
+ output_paths: list[Path] = []
+
+ for chunk_path in chunk_paths:
+ output_path = extraction_path_for_chunk(chunk_path, output_dir)
+ print(
+ f"Extracting {chunk_path.name} -> {output_path.name}",
+ flush=True,
+ )
+ extract_chunk(
+ chunk_path,
+ output_path,
+ model,
+ endpoint,
+ timeout,
+ temperature,
+ num_predict,
+ num_ctx,
+ )
+ output_paths.append(output_path)
+
+ return output_paths
+
+
+def main() -> int:
+ args = parse_args()
+
+ try:
+ output_paths = extract_input(
+ input_path=args.input,
+ output=args.output,
+ model=args.model,
+ endpoint=args.endpoint,
+ timeout=args.timeout,
+ temperature=args.temperature,
+ num_predict=args.num_predict,
+ num_ctx=args.num_ctx,
+ )
+ except requests.ConnectionError:
+ print(
+ "Error: Ollama is not reachable. Is `ollama serve` running?",
+ file=sys.stderr,
+ )
+ return 1
+ except requests.Timeout:
+ print("Error: The Ollama request timed out.", file=sys.stderr)
+ return 1
+ except requests.HTTPError as exc:
+ print(f"Error: Ollama returned an HTTP error: {exc}", file=sys.stderr)
+ return 1
+ except (OSError, UnicodeError, ValueError) as exc:
+ print(f"Error: {exc}", file=sys.stderr)
+ return 1
+
+ print(f"Input: {args.input}")
+ print(f"Processed chunks: {len(output_paths)}")
+ print(f"Extraction JSON files: {len(output_paths)}")
+ for output_path in output_paths:
+ print(output_path)
+
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/src/meeting_lab/normalization/normalize_transcript.py b/src/meeting_lab/normalization/normalize_transcript.py
index e69de29..ef4c584 100644
--- a/src/meeting_lab/normalization/normalize_transcript.py
+++ b/src/meeting_lab/normalization/normalize_transcript.py
@@ -0,0 +1,251 @@
+#!/usr/bin/env python3
+"""
+Conservatively normalize a meeting transcript without rewriting its meaning.
+
+The normalizer only performs low-risk cleanup:
+- removes isolated filler sounds such as "äh" and "ähm"
+- collapses immediate duplicate words or short duplicate phrases
+- normalizes whitespace
+- records every changed block in a JSON change log
+
+It deliberately does NOT remove modal words, qualifiers, negations,
+dates, numbers, responsibilities, technical statements, or commitments.
+
+Examples:
+ python normalize_transcript.py meeting_chunks/chunk_01.txt
+
+ python normalize_transcript.py meeting_chunks/chunk_01.txt \
+ --output meeting_chunks/chunk_01_normalized.txt \
+ --changes meeting_chunks/chunk_01_changes.json
+"""
+
+from __future__ import annotations
+
+import argparse
+import json
+import re
+import sys
+from dataclasses import asdict, dataclass
+from pathlib import Path
+
+
+# Only clearly non-semantic filler sounds.
+FILLER_PATTERN = re.compile(
+ r"(?i)(? "die Anlage"
+DUPLICATE_WORD_PATTERN = re.compile(
+ r"(?i)\b([A-Za-zÄÖÜäöüß][\wÄÖÜäöüß'-]*)"
+ r"(?:[\s,;:]+)\1\b"
+)
+
+# Immediate duplicate short phrase of two to four words:
+# "die Lüftung, die Lüftung" -> "die Lüftung"
+DUPLICATE_PHRASE_PATTERN = re.compile(
+ r"(?i)\b("
+ r"[A-Za-zÄÖÜäöüß][\wÄÖÜäöüß'-]*"
+ r"(?:\s+[A-Za-zÄÖÜäöüß][\wÄÖÜäöüß'-]*){1,3}"
+ r")"
+ r"(?:\s*[,;:]\s*|\s+)\1\b"
+)
+
+MULTISPACE_PATTERN = re.compile(r"[ \t]{2,}")
+SPACE_BEFORE_PUNCT_PATTERN = re.compile(r"\s+([,.;:!?])")
+MULTI_BLANK_PATTERN = re.compile(r"\n{3,}")
+
+
+@dataclass(frozen=True)
+class Change:
+ block_number: int
+ original: str
+ normalized: str
+ removed_fillers: list[str]
+ duplicate_reductions: list[str]
+
+
+def parse_args() -> argparse.Namespace:
+ parser = argparse.ArgumentParser(
+ description="Conservatively normalize a meeting transcript."
+ )
+ parser.add_argument("input_file", type=Path, help="Transcript text file")
+ parser.add_argument(
+ "-o",
+ "--output",
+ type=Path,
+ help="Normalized output file; default: _normalized.txt",
+ )
+ parser.add_argument(
+ "--changes",
+ type=Path,
+ help="Change log JSON; default: _changes.json",
+ )
+ return parser.parse_args()
+
+
+def split_blocks(text: str) -> list[str]:
+ """
+ Preserve transcript structure by treating blank-line-separated paragraphs
+ as atomic blocks. If there are no blank lines, use individual non-empty
+ lines as blocks.
+ """
+ normalized_newlines = text.replace("\r\n", "\n").replace("\r", "\n").strip()
+
+ paragraphs = [
+ part.strip()
+ for part in re.split(r"\n\s*\n+", normalized_newlines)
+ if part.strip()
+ ]
+ if len(paragraphs) > 1:
+ return paragraphs
+
+ return [line.strip() for line in normalized_newlines.splitlines() if line.strip()]
+
+
+def remove_fillers(text: str) -> tuple[str, list[str]]:
+ removed = [match.group(0) for match in FILLER_PATTERN.finditer(text)]
+ cleaned = FILLER_PATTERN.sub("", text)
+ return cleaned, removed
+
+
+def reduce_duplicate_phrases(text: str) -> tuple[str, list[str]]:
+ reductions: list[str] = []
+ current = text
+
+ # Repeat until stable because one correction may expose another.
+ for _ in range(5):
+ changed = False
+
+ def phrase_replacer(match: re.Match[str]) -> str:
+ nonlocal changed
+ changed = True
+ reductions.append(match.group(0))
+ return match.group(1)
+
+ updated = DUPLICATE_PHRASE_PATTERN.sub(phrase_replacer, current)
+ current = updated
+
+ def word_replacer(match: re.Match[str]) -> str:
+ nonlocal changed
+ changed = True
+ reductions.append(match.group(0))
+ return match.group(1)
+
+ updated = DUPLICATE_WORD_PATTERN.sub(word_replacer, current)
+ current = updated
+
+ if not changed:
+ break
+
+ return current, reductions
+
+
+def tidy_spacing(text: str) -> str:
+ text = MULTISPACE_PATTERN.sub(" ", text)
+ text = SPACE_BEFORE_PUNCT_PATTERN.sub(r"\1", text)
+ text = re.sub(r"([,;:])([^\s\n])", r"\1 \2", text)
+ text = re.sub(r"\(\s+", "(", text)
+ text = re.sub(r"\s+\)", ")", text)
+ return text.strip(" \t,;:")
+
+
+def normalize_block(block: str, block_number: int) -> tuple[str, Change | None]:
+ original = block
+
+ result, removed_fillers = remove_fillers(original)
+ result, duplicate_reductions = reduce_duplicate_phrases(result)
+ result = tidy_spacing(result)
+
+ if result == original:
+ return result, None
+
+ return result, Change(
+ block_number=block_number,
+ original=original,
+ normalized=result,
+ removed_fillers=removed_fillers,
+ duplicate_reductions=duplicate_reductions,
+ )
+
+
+def main() -> int:
+ args = parse_args()
+
+ try:
+ if not args.input_file.is_file():
+ raise FileNotFoundError(f"Input file not found: {args.input_file}")
+
+ source = args.input_file.read_text(encoding="utf-8-sig")
+ if not source.strip():
+ raise ValueError("The input file is empty.")
+
+ output_path = args.output or args.input_file.with_name(
+ f"{args.input_file.stem}_normalized.txt"
+ )
+ changes_path = args.changes or args.input_file.with_name(
+ f"{args.input_file.stem}_changes.json"
+ )
+
+ blocks = split_blocks(source)
+ normalized_blocks: list[str] = []
+ changes: list[Change] = []
+
+ for block_number, block in enumerate(blocks, start=1):
+ normalized, change = normalize_block(block, block_number)
+ normalized_blocks.append(normalized)
+ if change is not None:
+ changes.append(change)
+
+ normalized_text = "\n\n".join(normalized_blocks).strip() + "\n"
+
+ output_path.parent.mkdir(parents=True, exist_ok=True)
+ changes_path.parent.mkdir(parents=True, exist_ok=True)
+
+ output_path.write_text(normalized_text, encoding="utf-8")
+
+ report = {
+ "source_file": args.input_file.name,
+ "output_file": output_path.name,
+ "blocks_total": len(blocks),
+ "blocks_changed": len(changes),
+ "policy": {
+ "removed": [
+ "isolated filler sounds such as äh, ähm, hm, mhm",
+ "immediate duplicate words",
+ "immediate duplicate phrases of two to four words",
+ "redundant whitespace",
+ ],
+ "explicitly_preserved": [
+ "negations",
+ "modal words and qualifiers",
+ "dates, times, numbers and quantities",
+ "technical statements",
+ "responsibilities",
+ "deadlines",
+ "decisions and commitments",
+ ],
+ "principle": "When uncertain, leave the text unchanged.",
+ },
+ "changes": [asdict(change) for change in changes],
+ }
+
+ changes_path.write_text(
+ json.dumps(report, ensure_ascii=False, indent=2) + "\n",
+ encoding="utf-8",
+ )
+
+ print(f"Source: {args.input_file}")
+ print(f"Normalized: {output_path}")
+ print(f"Change log: {changes_path}")
+ print(f"Blocks total: {len(blocks)}")
+ print(f"Blocks changed: {len(changes)}")
+ return 0
+
+ except (OSError, UnicodeError, ValueError) as exc:
+ print(f"Error: {exc}", file=sys.stderr)
+ return 1
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/src/meeting_lab/protocol/build_protocol.py b/src/meeting_lab/protocol/build_protocol.py
index e69de29..f2f41bf 100644
--- a/src/meeting_lab/protocol/build_protocol.py
+++ b/src/meeting_lab/protocol/build_protocol.py
@@ -0,0 +1,223 @@
+#!/usr/bin/env python3
+"""
+Build a readable Markdown protocol from chunk extraction JSON files.
+"""
+
+from __future__ import annotations
+
+import argparse
+import json
+import re
+import sys
+from pathlib import Path
+from typing import Any
+
+
+EXTRACTION_CATEGORIES = (
+ "facts",
+ "decisions",
+ "todos",
+ "questions",
+ "positions",
+ "technical",
+)
+
+CATEGORY_TITLES = {
+ "facts": "Facts",
+ "decisions": "Decisions",
+ "todos": "Action Items",
+ "questions": "Open Questions",
+ "positions": "Positions",
+ "technical": "Technical",
+}
+
+CHUNK_NUMBER_RE = re.compile(r"chunk_(\d+)_")
+
+
+def parse_args() -> argparse.Namespace:
+ parser = argparse.ArgumentParser(
+ description="Build a Markdown meeting protocol from extraction JSON files."
+ )
+ parser.add_argument(
+ "input_dir",
+ type=Path,
+ help="Directory containing chunk_XX_extraction.json files",
+ )
+ parser.add_argument(
+ "-o",
+ "--output",
+ type=Path,
+ help="Output Markdown file; default: /meeting_protocol.md",
+ )
+ return parser.parse_args()
+
+
+def chunk_sort_key(path: Path) -> tuple[int, str]:
+ match = CHUNK_NUMBER_RE.search(path.name)
+ if not match:
+ return (sys.maxsize, path.name)
+ return (int(match.group(1)), path.name)
+
+
+def load_json(path: Path) -> Any:
+ return json.loads(path.read_text(encoding="utf-8-sig"))
+
+
+def find_extraction_files(input_dir: Path) -> list[Path]:
+ return sorted(
+ input_dir.glob("chunk_*_extraction.json"),
+ key=chunk_sort_key,
+ )
+
+
+def normalize_items(value: Any) -> list[str]:
+ if not isinstance(value, list):
+ return []
+
+ items: list[str] = []
+ for item in value:
+ if isinstance(item, str):
+ text = item.strip()
+ else:
+ text = json.dumps(item, ensure_ascii=False, sort_keys=True)
+ if text:
+ items.append(text)
+ return items
+
+
+def collect_extractions(
+ extraction_files: list[Path],
+) -> dict[str, list[str]]:
+ collected = {category: [] for category in EXTRACTION_CATEGORIES}
+
+ for path in extraction_files:
+ data = load_json(path)
+ if not isinstance(data, dict):
+ raise ValueError(
+ f"Extraction file must contain a JSON object: {path}"
+ )
+
+ for category in EXTRACTION_CATEGORIES:
+ collected[category].extend(
+ normalize_items(data.get(category, []))
+ )
+
+ return collected
+
+
+def find_topic_files(input_dir: Path) -> list[Path]:
+ return sorted(
+ input_dir.glob("chunk_*_normalized_windowed_segments.json"),
+ key=chunk_sort_key,
+ )
+
+
+def collect_topics(input_dir: Path) -> list[str]:
+ topics: list[str] = []
+
+ for path in find_topic_files(input_dir):
+ data = load_json(path)
+ if not isinstance(data, dict):
+ continue
+
+ source = data.get("source", {})
+ source_file = (
+ source.get("file")
+ if isinstance(source, dict)
+ else None
+ )
+ label = source_file or path.name
+
+ segments = data.get("segments", [])
+ if not isinstance(segments, list):
+ continue
+
+ for segment in segments:
+ if not isinstance(segment, dict):
+ continue
+ segment_id = segment.get("segment_id", "segment")
+ start_block = segment.get("start_block", "?")
+ end_block = segment.get("end_block", "?")
+ topics.append(
+ f"{label}: {segment_id} (blocks {start_block}-{end_block})"
+ )
+
+ return topics
+
+
+def render_list(items: list[str]) -> list[str]:
+ if not items:
+ return ["- None recorded."]
+ return [f"- {item}" for item in items]
+
+
+def build_protocol_markdown(
+ input_dir: Path,
+ extraction_files: list[Path],
+) -> str:
+ extractions = collect_extractions(extraction_files)
+ topics = collect_topics(input_dir)
+
+ lines = [
+ "# Meeting Protocol",
+ "",
+ "## Topics",
+ *render_list(topics),
+ "",
+ "## Facts",
+ *render_list(extractions["facts"]),
+ "",
+ "## Decisions",
+ *render_list(extractions["decisions"]),
+ "",
+ "## Open Questions",
+ *render_list(extractions["questions"]),
+ "",
+ "## Action Items",
+ *render_list(extractions["todos"]),
+ "",
+ "## Positions",
+ *render_list(extractions["positions"]),
+ "",
+ "## Technical",
+ *render_list(extractions["technical"]),
+ "",
+ ]
+ return "\n".join(lines)
+
+
+def build_protocol(input_dir: Path, output_path: Path | None = None) -> Path:
+ if not input_dir.is_dir():
+ raise FileNotFoundError(f"Input directory not found: {input_dir}")
+
+ extraction_files = find_extraction_files(input_dir)
+ if not extraction_files:
+ raise ValueError(
+ f"No extraction JSON files found in {input_dir}"
+ )
+
+ output = output_path or input_dir / "meeting_protocol.md"
+ markdown = build_protocol_markdown(input_dir, extraction_files)
+ output.write_text(markdown, encoding="utf-8")
+ return output
+
+
+def main() -> int:
+ args = parse_args()
+
+ try:
+ extraction_files = find_extraction_files(args.input_dir)
+ output_path = build_protocol(args.input_dir, args.output)
+ except (OSError, UnicodeError, ValueError, json.JSONDecodeError) as exc:
+ print(f"Error: {exc}", file=sys.stderr)
+ return 1
+
+ print(f"Input: {args.input_dir}")
+ print(f"Extraction JSON files: {len(extraction_files)}")
+ print(f"Protocol: {output_path}")
+
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/tests/test_extraction_protocol.py b/tests/test_extraction_protocol.py
new file mode 100644
index 0000000..dbadf9d
--- /dev/null
+++ b/tests/test_extraction_protocol.py
@@ -0,0 +1,124 @@
+import json
+import tempfile
+import unittest
+from pathlib import Path
+
+from src.meeting_lab.extraction.extract_chunks import (
+ EXTRACTION_CATEGORIES,
+ extraction_path_for_chunk,
+ normalize_current_schema,
+ parse_json_response,
+)
+from src.meeting_lab.protocol.build_protocol import build_protocol
+
+
+class ExtractionProtocolTests(unittest.TestCase):
+ def test_extraction_path_drops_normalized_suffix(self) -> None:
+ path = Path("chunks/chunk_01_normalized.txt")
+
+ self.assertEqual(
+ extraction_path_for_chunk(path),
+ Path("chunks/chunk_01_extraction.json"),
+ )
+
+ def test_normalize_current_schema_maps_legacy_response(self) -> None:
+ result = normalize_current_schema(
+ {
+ "facts": [
+ {
+ "speaker": "A",
+ "statement": "Fact one",
+ "status": "clear",
+ "evidence": "Fact",
+ }
+ ],
+ "decisions": [
+ {
+ "decision": "Decision one",
+ "evidence": "Decision",
+ }
+ ],
+ "todos": [
+ {
+ "task": "Todo one",
+ "responsible": "B",
+ "deadline": None,
+ "evidence": "Todo",
+ }
+ ],
+ "open_questions": [
+ {
+ "question": "Question one",
+ "evidence": "Question",
+ }
+ ],
+ "technical_details": [
+ {
+ "subject": "System",
+ "statement": "Technical one",
+ "status": "clear",
+ "evidence": "Technical",
+ }
+ ],
+ }
+ )
+
+ self.assertEqual(set(result), set(EXTRACTION_CATEGORIES))
+ self.assertEqual(result["positions"], [])
+ self.assertIn("Fact one", result["facts"][0])
+ self.assertIn("Decision one", result["decisions"][0])
+ self.assertIn("Todo one", result["todos"][0])
+ self.assertIn("Question one", result["questions"][0])
+ self.assertIn("Technical one", result["technical"][0])
+
+ def test_parse_json_response_uses_final_object_after_thinking(self) -> None:
+ text = """
+Thinking:
+I will reason about the transcript first.
+{"facts": ["draft"], "decisions": []}
+
+Final answer:
+{
+ "facts": ["final"],
+ "decisions": [],
+ "todos": [],
+ "questions": [],
+ "positions": [],
+ "technical": []
+}
+"""
+
+ parsed = parse_json_response(text)
+
+ self.assertEqual(parsed["facts"], ["final"])
+ self.assertEqual(set(parsed), set(EXTRACTION_CATEGORIES))
+
+ def test_build_protocol_groups_extraction_items(self) -> None:
+ with tempfile.TemporaryDirectory() as directory:
+ input_dir = Path(directory)
+ (input_dir / "chunk_01_extraction.json").write_text(
+ json.dumps(
+ {
+ "facts": ["Fact one"],
+ "decisions": ["Decision one"],
+ "todos": ["Todo one"],
+ "questions": ["Question one"],
+ "positions": [],
+ "technical": [],
+ }
+ ),
+ encoding="utf-8",
+ )
+
+ output_path = build_protocol(input_dir)
+
+ markdown = output_path.read_text(encoding="utf-8")
+ self.assertIn("# Meeting Protocol", markdown)
+ self.assertIn("## Facts\n- Fact one", markdown)
+ self.assertIn("## Decisions\n- Decision one", markdown)
+ self.assertIn("## Open Questions\n- Question one", markdown)
+ self.assertIn("## Action Items\n- Todo one", markdown)
+
+
+if __name__ == "__main__":
+ unittest.main()