From 07b0d80113fa1bb80dc4f3a2f912f67b1f1ec678 Mon Sep 17 00:00:00 2001 From: Martin Tazl Date: Thu, 30 Jul 2026 09:15:38 +0200 Subject: [PATCH] Implement first end-to-end meeting analysis pipeline --- .gitignore | 7 + scripts/clean_whisper_json.py | 139 +++++ src/meeting_lab/extraction/extract_chunks.py | 550 ++++++++++++++++++ .../normalization/normalize_transcript.py | 251 ++++++++ src/meeting_lab/protocol/build_protocol.py | 223 +++++++ tests/test_extraction_protocol.py | 124 ++++ 6 files changed, 1294 insertions(+) create mode 100644 scripts/clean_whisper_json.py create mode 100644 src/meeting_lab/extraction/extract_chunks.py create mode 100644 tests/test_extraction_protocol.py diff --git a/.gitignore b/.gitignore index aaffe1d..a516c1d 100644 --- a/.gitignore +++ b/.gitignore @@ -32,6 +32,13 @@ htmlcov/ experiments/**/output/ experiments/**/results/ +# Pipeline runtime artifacts +samples/raw/ +samples/whisper/ +**/chunk_*_extraction.json +**/meeting_protocol.md +**/*.raw.txt + # Lokale Meetings (niemals versionieren) meeting_data/ recordings/ diff --git a/scripts/clean_whisper_json.py b/scripts/clean_whisper_json.py new file mode 100644 index 0000000..a96f861 --- /dev/null +++ b/scripts/clean_whisper_json.py @@ -0,0 +1,139 @@ +import argparse +import json +import math +from pathlib import Path +from typing import Any + + +def is_nonfinite_number(value: Any) -> bool: + """Prüft auf NaN sowie positive oder negative Unendlichkeit.""" + return isinstance(value, float) and not math.isfinite(value) + + +def sanitize_nonfinite_values(value: Any) -> Any: + """ + Ersetzt NaN und Infinity rekursiv durch None. + + None wird in JSON als null geschrieben. + """ + if is_nonfinite_number(value): + return None + + if isinstance(value, dict): + return { + key: sanitize_nonfinite_values(item) + for key, item in value.items() + } + + if isinstance(value, list): + return [ + sanitize_nonfinite_values(item) + for item in value + ] + + return value + + +def is_invalid_empty_segment(segment: dict[str, Any]) -> bool: + """Erkennt technisch leere Whisper-Artefakte.""" + + text = str(segment.get("text", "")).strip() + start = segment.get("start") + end = segment.get("end") + avg_logprob = segment.get("avg_logprob") + + logprob_is_nan = ( + isinstance(avg_logprob, float) + and math.isnan(avg_logprob) + ) + + return ( + not text + and start == end + and logprob_is_nan + ) + + +def clean_whisper_json( + input_path: Path, + output_path: Path, +) -> None: + with input_path.open("r", encoding="utf-8") as file: + data = json.load(file) + + original_segments = data.get("segments", []) + + cleaned_segments = [ + segment + for segment in original_segments + if not is_invalid_empty_segment(segment) + ] + + for new_id, segment in enumerate(cleaned_segments): + segment["id"] = new_id + + data["segments"] = cleaned_segments + + data["text"] = " ".join( + str(segment.get("text", "")).strip() + for segment in cleaned_segments + if str(segment.get("text", "")).strip() + ) + + # Alle noch vorhandenen NaN-/Infinity-Werte durch null ersetzen. + data = sanitize_nonfinite_values(data) + + # Erst vollständig in einen String serialisieren. + # Dadurch bleibt keine unvollständige Ausgabedatei zurück, + # falls doch noch ein Fehler auftritt. + json_content = json.dumps( + data, + ensure_ascii=False, + indent=2, + allow_nan=False, + ) + + output_path.write_text( + json_content + "\n", + encoding="utf-8", + ) + + removed = len(original_segments) - len(cleaned_segments) + + print(f"Eingabedatei: {input_path}") + print(f"Ausgabedatei: {output_path}") + print(f"Segmente vorher: {len(original_segments)}") + print(f"Segmente nachher: {len(cleaned_segments)}") + print(f"Segmente entfernt: {removed}") + + +def main() -> None: + parser = argparse.ArgumentParser( + description=( + "Entfernt technisch leere Artefakte aus " + "einer MLX-Whisper-JSON-Datei." + ) + ) + parser.add_argument( + "input", + type=Path, + help="Whisper-JSON-Datei", + ) + parser.add_argument( + "-o", + "--output", + type=Path, + help="Ausgabedatei", + ) + + args = parser.parse_args() + + output_path = args.output or args.input.with_name( + f"{args.input.stem}_cleaned.json" + ) + + clean_whisper_json(args.input, output_path) + + +if __name__ == "__main__": + main() diff --git a/src/meeting_lab/extraction/extract_chunks.py b/src/meeting_lab/extraction/extract_chunks.py new file mode 100644 index 0000000..f51f6f0 --- /dev/null +++ b/src/meeting_lab/extraction/extract_chunks.py @@ -0,0 +1,550 @@ +#!/usr/bin/env python3 +""" +Extract structured meeting information from normalized transcript chunks. + +This module is adapted from /Users/tazl/quick-whisper-test/summarize.py. +It keeps the same Ollama extraction flow and conservative prompt rules, but +writes the current meeting-lab extraction schema. +""" + +from __future__ import annotations + +import argparse +import json +import re +import sys +import time +from pathlib import Path +from typing import Any + +import requests + + +DEFAULT_MODEL = "qwen3:8b" +DEFAULT_ENDPOINT = "http://localhost:11434/api/generate" + +EXTRACTION_CATEGORIES = ( + "facts", + "decisions", + "todos", + "questions", + "positions", + "technical", +) + +NORMALIZED_CHUNK_RE = re.compile(r"^(chunk_\d+)_normalized\.txt$") + + +SYSTEM_INSTRUCTION = """ +Du extrahierst Informationen aus Meeting-Transkripten. + +Arbeite ausschließlich mit dem vorgelegten Transkript. +Verwende kein eigenes Fachwissen, keine Vermutungen und keine üblichen +Funktionsweisen technischer Systeme. + +Regeln: +1. Erfinde nichts. +2. Interpretiere technische Aussagen nicht über den Wortlaut hinaus. +3. Korrigiere keine Aussagen anhand vermeintlichen Weltwissens. +4. Wenn etwas widersprüchlich oder unklar ist, kennzeichne es als unklar. +5. Übernimm wichtige technische Aussagen möglichst nah am Wortlaut. +6. Nenne bei Fakten nach Möglichkeit den Sprecher. +7. Ein Beschluss ist nur dann ein Beschluss, wenn im Text eine Einigung, + Freigabe oder verbindliche Festlegung erkennbar ist. +8. Eine Aufgabe ist nur dann eine Aufgabe, wenn eine Handlung und möglichst + eine verantwortliche Person oder Organisation erkennbar sind. +9. Gib ausschließlich gültiges JSON aus. Kein Markdown, keine Erläuterungen. +""".strip() + + +OUTPUT_SCHEMA = { + "chunk": { + "source_file": "string", + "summary": "Kurze, rein inhaltsbezogene Beschreibung des Chunks oder leer", + }, + "participants": ["string"], + "topics": ["string"], + "facts": [ + { + "speaker": "string oder null", + "statement": "Aussage möglichst nah am Wortlaut", + "status": "clear | unclear | contradictory", + "evidence": "Kurzes wörtliches oder nahezu wörtliches Textfragment", + } + ], + "decisions": [ + { + "decision": "string", + "evidence": "Kurzes Textfragment", + } + ], + "todos": [ + { + "task": "string", + "responsible": "string oder null", + "deadline": "string oder null", + "evidence": "Kurzes Textfragment", + } + ], + "open_questions": [ + { + "question": "string", + "evidence": "Kurzes Textfragment", + } + ], + "technical_details": [ + { + "subject": "string", + "statement": "Technische Aussage möglichst nah am Wortlaut", + "status": "clear | unclear | contradictory", + "evidence": "Kurzes Textfragment", + } + ], +} + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser( + description="Extract structured facts from normalized meeting chunks." + ) + parser.add_argument( + "input", + type=Path, + help=( + "A chunk_XX_normalized.txt file or a directory containing " + "normalized chunk files" + ), + ) + parser.add_argument( + "-o", + "--output", + type=Path, + help=( + "Output JSON path for a single input file, or output directory " + "for a directory input" + ), + ) + parser.add_argument( + "--model", + default=DEFAULT_MODEL, + help=f"Ollama model name (default: {DEFAULT_MODEL})", + ) + parser.add_argument( + "--endpoint", + default=DEFAULT_ENDPOINT, + help=f"Ollama generate endpoint (default: {DEFAULT_ENDPOINT})", + ) + parser.add_argument( + "--timeout", + type=int, + default=1800, + help="HTTP timeout in seconds per chunk (default: 1800)", + ) + parser.add_argument( + "--temperature", + type=float, + default=0.0, + help="Sampling temperature (default: 0.0)", + ) + parser.add_argument( + "--num-predict", + type=int, + default=8192, + help="Maximum generated tokens per chunk (default: 8192)", + ) + parser.add_argument( + "--num-ctx", + type=int, + default=32768, + help="Context window tokens per chunk (default: 32768)", + ) + return parser.parse_args() + + +def build_prompt(source_name: str, transcript: str) -> str: + schema_text = json.dumps(OUTPUT_SCHEMA, ensure_ascii=False, indent=2) + + return f"""{SYSTEM_INSTRUCTION} + +Quelldatei: +{source_name} + +Erwartete JSON-Struktur: +{schema_text} + +Hinweise zur Ausgabe: +- Alle obersten Schlüssel müssen vorhanden sein. +- Verwende leere Listen, wenn keine Einträge vorhanden sind. +- Verwende null, wenn Verantwortliche, Sprecher oder Termine nicht erkennbar sind. +- "evidence" muss sich eng am Transkript orientieren. +- Ersetze technische Aussagen niemals durch eine vermeintlich korrektere Erklärung. +- Confidence-Werte sind ausdrücklich nicht erwünscht. + +TRANSKRIPT: +--- BEGINN TRANSKRIPT --- +{transcript} +--- ENDE TRANSKRIPT --- +""" + + +def call_ollama( + endpoint: str, + model: str, + prompt: str, + timeout: int, + temperature: float, + num_predict: int | None, + num_ctx: int | None, +) -> tuple[str, dict[str, Any]]: + options: dict[str, Any] = { + "temperature": temperature, + } + if num_predict is not None: + options["num_predict"] = num_predict + if num_ctx is not None: + options["num_ctx"] = num_ctx + + payload = { + "model": model, + "prompt": prompt, + "think": False, + "stream": False, + "format": "json", + "options": options, + } + + started = time.perf_counter() + response = requests.post(endpoint, json=payload, timeout=timeout) + elapsed = time.perf_counter() - started + response.raise_for_status() + + data = response.json() + text = response_text_from_ollama_data(data) + if not isinstance(text, str) or not text.strip(): + raise ValueError("Ollama returned no usable response text.") + + metadata = { + "model": data.get("model", model), + "elapsed_seconds": round(elapsed, 3), + "total_duration_ns": data.get("total_duration"), + "load_duration_ns": data.get("load_duration"), + "prompt_eval_count": data.get("prompt_eval_count"), + "prompt_eval_duration_ns": data.get("prompt_eval_duration"), + "eval_count": data.get("eval_count"), + "eval_duration_ns": data.get("eval_duration"), + } + return text.strip(), metadata + + +def response_text_from_ollama_data(data: dict[str, Any]) -> str | None: + text = data.get("response") + if isinstance(text, str) and text.strip(): + return text + + message = data.get("message") + if isinstance(message, dict): + content = message.get("content") + if isinstance(content, str) and content.strip(): + return content + + thinking = data.get("thinking") + if isinstance(thinking, str) and thinking.strip(): + return thinking + + return text if isinstance(text, str) else None + + +def iter_json_object_candidates(text: str) -> list[str]: + candidates: list[str] = [] + start: int | None = None + depth = 0 + in_string = False + escaped = False + + for index, char in enumerate(text): + if in_string: + if escaped: + escaped = False + elif char == "\\": + escaped = True + elif char == '"': + in_string = False + continue + + if char == '"': + in_string = True + continue + + if char == "{": + if depth == 0: + start = index + depth += 1 + continue + + if char == "}" and depth: + depth -= 1 + if depth == 0 and start is not None: + candidates.append(text[start : index + 1]) + start = None + + return candidates + + +def parse_json_response(text: str) -> dict[str, Any]: + try: + parsed = json.loads(text) + except json.JSONDecodeError: + parsed = None + for candidate in iter_json_object_candidates(text): + try: + candidate_json = json.loads(candidate) + except json.JSONDecodeError: + continue + if isinstance(candidate_json, dict): + parsed = candidate_json + if parsed is None: + raise + + if not isinstance(parsed, dict): + raise ValueError("The model response is valid JSON but not a JSON object.") + return parsed + + +def text_from_value(value: Any, preferred_keys: tuple[str, ...]) -> str: + if isinstance(value, str): + return value.strip() + + if isinstance(value, dict): + parts: list[str] = [] + for key in preferred_keys: + item = value.get(key) + if item is None: + continue + text = str(item).strip() + if text: + parts.append(text) + if parts: + return " | ".join(parts) + return json.dumps(value, ensure_ascii=False, sort_keys=True) + + return str(value).strip() + + +def normalize_current_schema(result: dict[str, Any]) -> dict[str, list[str]]: + return { + "facts": [ + text + for item in result.get("facts", []) + if ( + text := text_from_value( + item, + ("speaker", "statement", "status", "evidence"), + ) + ) + ], + "decisions": [ + text + for item in result.get("decisions", []) + if ( + text := text_from_value( + item, + ("decision", "evidence"), + ) + ) + ], + "todos": [ + text + for item in result.get("todos", []) + if ( + text := text_from_value( + item, + ("task", "responsible", "deadline", "evidence"), + ) + ) + ], + "questions": [ + text + for item in result.get("open_questions", result.get("questions", [])) + if ( + text := text_from_value( + item, + ("question", "evidence"), + ) + ) + ], + "positions": [ + text + for item in result.get("positions", []) + if ( + text := text_from_value( + item, + ("speaker", "position", "statement", "evidence"), + ) + ) + ], + "technical": [ + text + for item in result.get("technical_details", result.get("technical", [])) + if ( + text := text_from_value( + item, + ("subject", "statement", "status", "evidence"), + ) + ) + ], + } + + +def extraction_path_for_chunk(chunk_path: Path, output_dir: Path | None = None) -> Path: + match = NORMALIZED_CHUNK_RE.match(chunk_path.name) + if not match: + raise ValueError( + f"Not a normalized chunk filename: {chunk_path.name}" + ) + directory = output_dir or chunk_path.parent + return directory / f"{match.group(1)}_extraction.json" + + +def find_normalized_chunks(input_dir: Path) -> list[Path]: + return sorted(input_dir.glob("chunk_*_normalized.txt")) + + +def extract_chunk( + chunk_path: Path, + output_path: Path, + model: str, + endpoint: str, + timeout: int, + temperature: float, + num_predict: int | None, + num_ctx: int | None, +) -> dict[str, list[str]]: + transcript = chunk_path.read_text(encoding="utf-8-sig").strip() + if not transcript: + raise ValueError(f"The input file is empty: {chunk_path}") + + prompt = build_prompt(chunk_path.name, transcript) + raw_text, _metadata = call_ollama( + endpoint=endpoint, + model=model, + prompt=prompt, + timeout=timeout, + temperature=temperature, + num_predict=num_predict, + num_ctx=num_ctx, + ) + + try: + parsed = parse_json_response(raw_text) + except (json.JSONDecodeError, ValueError) as exc: + raw_path = output_path.with_suffix(".raw.txt") + raw_path.write_text(raw_text + "\n", encoding="utf-8") + raise ValueError( + f"Model output was not valid JSON. Raw output saved to: {raw_path}" + ) from exc + + extraction = normalize_current_schema(parsed) + output_path.parent.mkdir(parents=True, exist_ok=True) + output_path.write_text( + json.dumps(extraction, ensure_ascii=False, indent=2) + "\n", + encoding="utf-8", + ) + return extraction + + +def extract_input( + input_path: Path, + output: Path | None, + model: str, + endpoint: str, + timeout: int, + temperature: float, + num_predict: int | None, + num_ctx: int | None, +) -> list[Path]: + if input_path.is_file(): + output_path = output or extraction_path_for_chunk(input_path) + extract_chunk( + input_path, + output_path, + model, + endpoint, + timeout, + temperature, + num_predict, + num_ctx, + ) + return [output_path] + + if not input_path.is_dir(): + raise FileNotFoundError(f"Input path not found: {input_path}") + + chunk_paths = find_normalized_chunks(input_path) + if not chunk_paths: + raise ValueError(f"No normalized chunks found in {input_path}") + + output_dir = output if output is not None else input_path + output_paths: list[Path] = [] + + for chunk_path in chunk_paths: + output_path = extraction_path_for_chunk(chunk_path, output_dir) + print( + f"Extracting {chunk_path.name} -> {output_path.name}", + flush=True, + ) + extract_chunk( + chunk_path, + output_path, + model, + endpoint, + timeout, + temperature, + num_predict, + num_ctx, + ) + output_paths.append(output_path) + + return output_paths + + +def main() -> int: + args = parse_args() + + try: + output_paths = extract_input( + input_path=args.input, + output=args.output, + model=args.model, + endpoint=args.endpoint, + timeout=args.timeout, + temperature=args.temperature, + num_predict=args.num_predict, + num_ctx=args.num_ctx, + ) + except requests.ConnectionError: + print( + "Error: Ollama is not reachable. Is `ollama serve` running?", + file=sys.stderr, + ) + return 1 + except requests.Timeout: + print("Error: The Ollama request timed out.", file=sys.stderr) + return 1 + except requests.HTTPError as exc: + print(f"Error: Ollama returned an HTTP error: {exc}", file=sys.stderr) + return 1 + except (OSError, UnicodeError, ValueError) as exc: + print(f"Error: {exc}", file=sys.stderr) + return 1 + + print(f"Input: {args.input}") + print(f"Processed chunks: {len(output_paths)}") + print(f"Extraction JSON files: {len(output_paths)}") + for output_path in output_paths: + print(output_path) + + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/meeting_lab/normalization/normalize_transcript.py b/src/meeting_lab/normalization/normalize_transcript.py index e69de29..ef4c584 100644 --- a/src/meeting_lab/normalization/normalize_transcript.py +++ b/src/meeting_lab/normalization/normalize_transcript.py @@ -0,0 +1,251 @@ +#!/usr/bin/env python3 +""" +Conservatively normalize a meeting transcript without rewriting its meaning. + +The normalizer only performs low-risk cleanup: +- removes isolated filler sounds such as "äh" and "ähm" +- collapses immediate duplicate words or short duplicate phrases +- normalizes whitespace +- records every changed block in a JSON change log + +It deliberately does NOT remove modal words, qualifiers, negations, +dates, numbers, responsibilities, technical statements, or commitments. + +Examples: + python normalize_transcript.py meeting_chunks/chunk_01.txt + + python normalize_transcript.py meeting_chunks/chunk_01.txt \ + --output meeting_chunks/chunk_01_normalized.txt \ + --changes meeting_chunks/chunk_01_changes.json +""" + +from __future__ import annotations + +import argparse +import json +import re +import sys +from dataclasses import asdict, dataclass +from pathlib import Path + + +# Only clearly non-semantic filler sounds. +FILLER_PATTERN = re.compile( + r"(?i)(? "die Anlage" +DUPLICATE_WORD_PATTERN = re.compile( + r"(?i)\b([A-Za-zÄÖÜäöüß][\wÄÖÜäöüß'-]*)" + r"(?:[\s,;:]+)\1\b" +) + +# Immediate duplicate short phrase of two to four words: +# "die Lüftung, die Lüftung" -> "die Lüftung" +DUPLICATE_PHRASE_PATTERN = re.compile( + r"(?i)\b(" + r"[A-Za-zÄÖÜäöüß][\wÄÖÜäöüß'-]*" + r"(?:\s+[A-Za-zÄÖÜäöüß][\wÄÖÜäöüß'-]*){1,3}" + r")" + r"(?:\s*[,;:]\s*|\s+)\1\b" +) + +MULTISPACE_PATTERN = re.compile(r"[ \t]{2,}") +SPACE_BEFORE_PUNCT_PATTERN = re.compile(r"\s+([,.;:!?])") +MULTI_BLANK_PATTERN = re.compile(r"\n{3,}") + + +@dataclass(frozen=True) +class Change: + block_number: int + original: str + normalized: str + removed_fillers: list[str] + duplicate_reductions: list[str] + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser( + description="Conservatively normalize a meeting transcript." + ) + parser.add_argument("input_file", type=Path, help="Transcript text file") + parser.add_argument( + "-o", + "--output", + type=Path, + help="Normalized output file; default: _normalized.txt", + ) + parser.add_argument( + "--changes", + type=Path, + help="Change log JSON; default: _changes.json", + ) + return parser.parse_args() + + +def split_blocks(text: str) -> list[str]: + """ + Preserve transcript structure by treating blank-line-separated paragraphs + as atomic blocks. If there are no blank lines, use individual non-empty + lines as blocks. + """ + normalized_newlines = text.replace("\r\n", "\n").replace("\r", "\n").strip() + + paragraphs = [ + part.strip() + for part in re.split(r"\n\s*\n+", normalized_newlines) + if part.strip() + ] + if len(paragraphs) > 1: + return paragraphs + + return [line.strip() for line in normalized_newlines.splitlines() if line.strip()] + + +def remove_fillers(text: str) -> tuple[str, list[str]]: + removed = [match.group(0) for match in FILLER_PATTERN.finditer(text)] + cleaned = FILLER_PATTERN.sub("", text) + return cleaned, removed + + +def reduce_duplicate_phrases(text: str) -> tuple[str, list[str]]: + reductions: list[str] = [] + current = text + + # Repeat until stable because one correction may expose another. + for _ in range(5): + changed = False + + def phrase_replacer(match: re.Match[str]) -> str: + nonlocal changed + changed = True + reductions.append(match.group(0)) + return match.group(1) + + updated = DUPLICATE_PHRASE_PATTERN.sub(phrase_replacer, current) + current = updated + + def word_replacer(match: re.Match[str]) -> str: + nonlocal changed + changed = True + reductions.append(match.group(0)) + return match.group(1) + + updated = DUPLICATE_WORD_PATTERN.sub(word_replacer, current) + current = updated + + if not changed: + break + + return current, reductions + + +def tidy_spacing(text: str) -> str: + text = MULTISPACE_PATTERN.sub(" ", text) + text = SPACE_BEFORE_PUNCT_PATTERN.sub(r"\1", text) + text = re.sub(r"([,;:])([^\s\n])", r"\1 \2", text) + text = re.sub(r"\(\s+", "(", text) + text = re.sub(r"\s+\)", ")", text) + return text.strip(" \t,;:") + + +def normalize_block(block: str, block_number: int) -> tuple[str, Change | None]: + original = block + + result, removed_fillers = remove_fillers(original) + result, duplicate_reductions = reduce_duplicate_phrases(result) + result = tidy_spacing(result) + + if result == original: + return result, None + + return result, Change( + block_number=block_number, + original=original, + normalized=result, + removed_fillers=removed_fillers, + duplicate_reductions=duplicate_reductions, + ) + + +def main() -> int: + args = parse_args() + + try: + if not args.input_file.is_file(): + raise FileNotFoundError(f"Input file not found: {args.input_file}") + + source = args.input_file.read_text(encoding="utf-8-sig") + if not source.strip(): + raise ValueError("The input file is empty.") + + output_path = args.output or args.input_file.with_name( + f"{args.input_file.stem}_normalized.txt" + ) + changes_path = args.changes or args.input_file.with_name( + f"{args.input_file.stem}_changes.json" + ) + + blocks = split_blocks(source) + normalized_blocks: list[str] = [] + changes: list[Change] = [] + + for block_number, block in enumerate(blocks, start=1): + normalized, change = normalize_block(block, block_number) + normalized_blocks.append(normalized) + if change is not None: + changes.append(change) + + normalized_text = "\n\n".join(normalized_blocks).strip() + "\n" + + output_path.parent.mkdir(parents=True, exist_ok=True) + changes_path.parent.mkdir(parents=True, exist_ok=True) + + output_path.write_text(normalized_text, encoding="utf-8") + + report = { + "source_file": args.input_file.name, + "output_file": output_path.name, + "blocks_total": len(blocks), + "blocks_changed": len(changes), + "policy": { + "removed": [ + "isolated filler sounds such as äh, ähm, hm, mhm", + "immediate duplicate words", + "immediate duplicate phrases of two to four words", + "redundant whitespace", + ], + "explicitly_preserved": [ + "negations", + "modal words and qualifiers", + "dates, times, numbers and quantities", + "technical statements", + "responsibilities", + "deadlines", + "decisions and commitments", + ], + "principle": "When uncertain, leave the text unchanged.", + }, + "changes": [asdict(change) for change in changes], + } + + changes_path.write_text( + json.dumps(report, ensure_ascii=False, indent=2) + "\n", + encoding="utf-8", + ) + + print(f"Source: {args.input_file}") + print(f"Normalized: {output_path}") + print(f"Change log: {changes_path}") + print(f"Blocks total: {len(blocks)}") + print(f"Blocks changed: {len(changes)}") + return 0 + + except (OSError, UnicodeError, ValueError) as exc: + print(f"Error: {exc}", file=sys.stderr) + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/meeting_lab/protocol/build_protocol.py b/src/meeting_lab/protocol/build_protocol.py index e69de29..f2f41bf 100644 --- a/src/meeting_lab/protocol/build_protocol.py +++ b/src/meeting_lab/protocol/build_protocol.py @@ -0,0 +1,223 @@ +#!/usr/bin/env python3 +""" +Build a readable Markdown protocol from chunk extraction JSON files. +""" + +from __future__ import annotations + +import argparse +import json +import re +import sys +from pathlib import Path +from typing import Any + + +EXTRACTION_CATEGORIES = ( + "facts", + "decisions", + "todos", + "questions", + "positions", + "technical", +) + +CATEGORY_TITLES = { + "facts": "Facts", + "decisions": "Decisions", + "todos": "Action Items", + "questions": "Open Questions", + "positions": "Positions", + "technical": "Technical", +} + +CHUNK_NUMBER_RE = re.compile(r"chunk_(\d+)_") + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser( + description="Build a Markdown meeting protocol from extraction JSON files." + ) + parser.add_argument( + "input_dir", + type=Path, + help="Directory containing chunk_XX_extraction.json files", + ) + parser.add_argument( + "-o", + "--output", + type=Path, + help="Output Markdown file; default: /meeting_protocol.md", + ) + return parser.parse_args() + + +def chunk_sort_key(path: Path) -> tuple[int, str]: + match = CHUNK_NUMBER_RE.search(path.name) + if not match: + return (sys.maxsize, path.name) + return (int(match.group(1)), path.name) + + +def load_json(path: Path) -> Any: + return json.loads(path.read_text(encoding="utf-8-sig")) + + +def find_extraction_files(input_dir: Path) -> list[Path]: + return sorted( + input_dir.glob("chunk_*_extraction.json"), + key=chunk_sort_key, + ) + + +def normalize_items(value: Any) -> list[str]: + if not isinstance(value, list): + return [] + + items: list[str] = [] + for item in value: + if isinstance(item, str): + text = item.strip() + else: + text = json.dumps(item, ensure_ascii=False, sort_keys=True) + if text: + items.append(text) + return items + + +def collect_extractions( + extraction_files: list[Path], +) -> dict[str, list[str]]: + collected = {category: [] for category in EXTRACTION_CATEGORIES} + + for path in extraction_files: + data = load_json(path) + if not isinstance(data, dict): + raise ValueError( + f"Extraction file must contain a JSON object: {path}" + ) + + for category in EXTRACTION_CATEGORIES: + collected[category].extend( + normalize_items(data.get(category, [])) + ) + + return collected + + +def find_topic_files(input_dir: Path) -> list[Path]: + return sorted( + input_dir.glob("chunk_*_normalized_windowed_segments.json"), + key=chunk_sort_key, + ) + + +def collect_topics(input_dir: Path) -> list[str]: + topics: list[str] = [] + + for path in find_topic_files(input_dir): + data = load_json(path) + if not isinstance(data, dict): + continue + + source = data.get("source", {}) + source_file = ( + source.get("file") + if isinstance(source, dict) + else None + ) + label = source_file or path.name + + segments = data.get("segments", []) + if not isinstance(segments, list): + continue + + for segment in segments: + if not isinstance(segment, dict): + continue + segment_id = segment.get("segment_id", "segment") + start_block = segment.get("start_block", "?") + end_block = segment.get("end_block", "?") + topics.append( + f"{label}: {segment_id} (blocks {start_block}-{end_block})" + ) + + return topics + + +def render_list(items: list[str]) -> list[str]: + if not items: + return ["- None recorded."] + return [f"- {item}" for item in items] + + +def build_protocol_markdown( + input_dir: Path, + extraction_files: list[Path], +) -> str: + extractions = collect_extractions(extraction_files) + topics = collect_topics(input_dir) + + lines = [ + "# Meeting Protocol", + "", + "## Topics", + *render_list(topics), + "", + "## Facts", + *render_list(extractions["facts"]), + "", + "## Decisions", + *render_list(extractions["decisions"]), + "", + "## Open Questions", + *render_list(extractions["questions"]), + "", + "## Action Items", + *render_list(extractions["todos"]), + "", + "## Positions", + *render_list(extractions["positions"]), + "", + "## Technical", + *render_list(extractions["technical"]), + "", + ] + return "\n".join(lines) + + +def build_protocol(input_dir: Path, output_path: Path | None = None) -> Path: + if not input_dir.is_dir(): + raise FileNotFoundError(f"Input directory not found: {input_dir}") + + extraction_files = find_extraction_files(input_dir) + if not extraction_files: + raise ValueError( + f"No extraction JSON files found in {input_dir}" + ) + + output = output_path or input_dir / "meeting_protocol.md" + markdown = build_protocol_markdown(input_dir, extraction_files) + output.write_text(markdown, encoding="utf-8") + return output + + +def main() -> int: + args = parse_args() + + try: + extraction_files = find_extraction_files(args.input_dir) + output_path = build_protocol(args.input_dir, args.output) + except (OSError, UnicodeError, ValueError, json.JSONDecodeError) as exc: + print(f"Error: {exc}", file=sys.stderr) + return 1 + + print(f"Input: {args.input_dir}") + print(f"Extraction JSON files: {len(extraction_files)}") + print(f"Protocol: {output_path}") + + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_extraction_protocol.py b/tests/test_extraction_protocol.py new file mode 100644 index 0000000..dbadf9d --- /dev/null +++ b/tests/test_extraction_protocol.py @@ -0,0 +1,124 @@ +import json +import tempfile +import unittest +from pathlib import Path + +from src.meeting_lab.extraction.extract_chunks import ( + EXTRACTION_CATEGORIES, + extraction_path_for_chunk, + normalize_current_schema, + parse_json_response, +) +from src.meeting_lab.protocol.build_protocol import build_protocol + + +class ExtractionProtocolTests(unittest.TestCase): + def test_extraction_path_drops_normalized_suffix(self) -> None: + path = Path("chunks/chunk_01_normalized.txt") + + self.assertEqual( + extraction_path_for_chunk(path), + Path("chunks/chunk_01_extraction.json"), + ) + + def test_normalize_current_schema_maps_legacy_response(self) -> None: + result = normalize_current_schema( + { + "facts": [ + { + "speaker": "A", + "statement": "Fact one", + "status": "clear", + "evidence": "Fact", + } + ], + "decisions": [ + { + "decision": "Decision one", + "evidence": "Decision", + } + ], + "todos": [ + { + "task": "Todo one", + "responsible": "B", + "deadline": None, + "evidence": "Todo", + } + ], + "open_questions": [ + { + "question": "Question one", + "evidence": "Question", + } + ], + "technical_details": [ + { + "subject": "System", + "statement": "Technical one", + "status": "clear", + "evidence": "Technical", + } + ], + } + ) + + self.assertEqual(set(result), set(EXTRACTION_CATEGORIES)) + self.assertEqual(result["positions"], []) + self.assertIn("Fact one", result["facts"][0]) + self.assertIn("Decision one", result["decisions"][0]) + self.assertIn("Todo one", result["todos"][0]) + self.assertIn("Question one", result["questions"][0]) + self.assertIn("Technical one", result["technical"][0]) + + def test_parse_json_response_uses_final_object_after_thinking(self) -> None: + text = """ +Thinking: +I will reason about the transcript first. +{"facts": ["draft"], "decisions": []} + +Final answer: +{ + "facts": ["final"], + "decisions": [], + "todos": [], + "questions": [], + "positions": [], + "technical": [] +} +""" + + parsed = parse_json_response(text) + + self.assertEqual(parsed["facts"], ["final"]) + self.assertEqual(set(parsed), set(EXTRACTION_CATEGORIES)) + + def test_build_protocol_groups_extraction_items(self) -> None: + with tempfile.TemporaryDirectory() as directory: + input_dir = Path(directory) + (input_dir / "chunk_01_extraction.json").write_text( + json.dumps( + { + "facts": ["Fact one"], + "decisions": ["Decision one"], + "todos": ["Todo one"], + "questions": ["Question one"], + "positions": [], + "technical": [], + } + ), + encoding="utf-8", + ) + + output_path = build_protocol(input_dir) + + markdown = output_path.read_text(encoding="utf-8") + self.assertIn("# Meeting Protocol", markdown) + self.assertIn("## Facts\n- Fact one", markdown) + self.assertIn("## Decisions\n- Decision one", markdown) + self.assertIn("## Open Questions\n- Question one", markdown) + self.assertIn("## Action Items\n- Todo one", markdown) + + +if __name__ == "__main__": + unittest.main()