Implement first end-to-end meeting analysis pipeline

This commit is contained in:
2026-07-30 09:15:38 +02:00
parent 46565233d8
commit 07b0d80113
6 changed files with 1294 additions and 0 deletions
+7
View File
@@ -32,6 +32,13 @@ htmlcov/
experiments/**/output/
experiments/**/results/
# Pipeline runtime artifacts
samples/raw/
samples/whisper/
**/chunk_*_extraction.json
**/meeting_protocol.md
**/*.raw.txt
# Lokale Meetings (niemals versionieren)
meeting_data/
recordings/
+139
View File
@@ -0,0 +1,139 @@
import argparse
import json
import math
from pathlib import Path
from typing import Any
def is_nonfinite_number(value: Any) -> bool:
"""Prüft auf NaN sowie positive oder negative Unendlichkeit."""
return isinstance(value, float) and not math.isfinite(value)
def sanitize_nonfinite_values(value: Any) -> Any:
"""
Ersetzt NaN und Infinity rekursiv durch None.
None wird in JSON als null geschrieben.
"""
if is_nonfinite_number(value):
return None
if isinstance(value, dict):
return {
key: sanitize_nonfinite_values(item)
for key, item in value.items()
}
if isinstance(value, list):
return [
sanitize_nonfinite_values(item)
for item in value
]
return value
def is_invalid_empty_segment(segment: dict[str, Any]) -> bool:
"""Erkennt technisch leere Whisper-Artefakte."""
text = str(segment.get("text", "")).strip()
start = segment.get("start")
end = segment.get("end")
avg_logprob = segment.get("avg_logprob")
logprob_is_nan = (
isinstance(avg_logprob, float)
and math.isnan(avg_logprob)
)
return (
not text
and start == end
and logprob_is_nan
)
def clean_whisper_json(
input_path: Path,
output_path: Path,
) -> None:
with input_path.open("r", encoding="utf-8") as file:
data = json.load(file)
original_segments = data.get("segments", [])
cleaned_segments = [
segment
for segment in original_segments
if not is_invalid_empty_segment(segment)
]
for new_id, segment in enumerate(cleaned_segments):
segment["id"] = new_id
data["segments"] = cleaned_segments
data["text"] = " ".join(
str(segment.get("text", "")).strip()
for segment in cleaned_segments
if str(segment.get("text", "")).strip()
)
# Alle noch vorhandenen NaN-/Infinity-Werte durch null ersetzen.
data = sanitize_nonfinite_values(data)
# Erst vollständig in einen String serialisieren.
# Dadurch bleibt keine unvollständige Ausgabedatei zurück,
# falls doch noch ein Fehler auftritt.
json_content = json.dumps(
data,
ensure_ascii=False,
indent=2,
allow_nan=False,
)
output_path.write_text(
json_content + "\n",
encoding="utf-8",
)
removed = len(original_segments) - len(cleaned_segments)
print(f"Eingabedatei: {input_path}")
print(f"Ausgabedatei: {output_path}")
print(f"Segmente vorher: {len(original_segments)}")
print(f"Segmente nachher: {len(cleaned_segments)}")
print(f"Segmente entfernt: {removed}")
def main() -> None:
parser = argparse.ArgumentParser(
description=(
"Entfernt technisch leere Artefakte aus "
"einer MLX-Whisper-JSON-Datei."
)
)
parser.add_argument(
"input",
type=Path,
help="Whisper-JSON-Datei",
)
parser.add_argument(
"-o",
"--output",
type=Path,
help="Ausgabedatei",
)
args = parser.parse_args()
output_path = args.output or args.input.with_name(
f"{args.input.stem}_cleaned.json"
)
clean_whisper_json(args.input, output_path)
if __name__ == "__main__":
main()
@@ -0,0 +1,550 @@
#!/usr/bin/env python3
"""
Extract structured meeting information from normalized transcript chunks.
This module is adapted from /Users/tazl/quick-whisper-test/summarize.py.
It keeps the same Ollama extraction flow and conservative prompt rules, but
writes the current meeting-lab extraction schema.
"""
from __future__ import annotations
import argparse
import json
import re
import sys
import time
from pathlib import Path
from typing import Any
import requests
DEFAULT_MODEL = "qwen3:8b"
DEFAULT_ENDPOINT = "http://localhost:11434/api/generate"
EXTRACTION_CATEGORIES = (
"facts",
"decisions",
"todos",
"questions",
"positions",
"technical",
)
NORMALIZED_CHUNK_RE = re.compile(r"^(chunk_\d+)_normalized\.txt$")
SYSTEM_INSTRUCTION = """
Du extrahierst Informationen aus Meeting-Transkripten.
Arbeite ausschließlich mit dem vorgelegten Transkript.
Verwende kein eigenes Fachwissen, keine Vermutungen und keine üblichen
Funktionsweisen technischer Systeme.
Regeln:
1. Erfinde nichts.
2. Interpretiere technische Aussagen nicht über den Wortlaut hinaus.
3. Korrigiere keine Aussagen anhand vermeintlichen Weltwissens.
4. Wenn etwas widersprüchlich oder unklar ist, kennzeichne es als unklar.
5. Übernimm wichtige technische Aussagen möglichst nah am Wortlaut.
6. Nenne bei Fakten nach Möglichkeit den Sprecher.
7. Ein Beschluss ist nur dann ein Beschluss, wenn im Text eine Einigung,
Freigabe oder verbindliche Festlegung erkennbar ist.
8. Eine Aufgabe ist nur dann eine Aufgabe, wenn eine Handlung und möglichst
eine verantwortliche Person oder Organisation erkennbar sind.
9. Gib ausschließlich gültiges JSON aus. Kein Markdown, keine Erläuterungen.
""".strip()
OUTPUT_SCHEMA = {
"chunk": {
"source_file": "string",
"summary": "Kurze, rein inhaltsbezogene Beschreibung des Chunks oder leer",
},
"participants": ["string"],
"topics": ["string"],
"facts": [
{
"speaker": "string oder null",
"statement": "Aussage möglichst nah am Wortlaut",
"status": "clear | unclear | contradictory",
"evidence": "Kurzes wörtliches oder nahezu wörtliches Textfragment",
}
],
"decisions": [
{
"decision": "string",
"evidence": "Kurzes Textfragment",
}
],
"todos": [
{
"task": "string",
"responsible": "string oder null",
"deadline": "string oder null",
"evidence": "Kurzes Textfragment",
}
],
"open_questions": [
{
"question": "string",
"evidence": "Kurzes Textfragment",
}
],
"technical_details": [
{
"subject": "string",
"statement": "Technische Aussage möglichst nah am Wortlaut",
"status": "clear | unclear | contradictory",
"evidence": "Kurzes Textfragment",
}
],
}
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Extract structured facts from normalized meeting chunks."
)
parser.add_argument(
"input",
type=Path,
help=(
"A chunk_XX_normalized.txt file or a directory containing "
"normalized chunk files"
),
)
parser.add_argument(
"-o",
"--output",
type=Path,
help=(
"Output JSON path for a single input file, or output directory "
"for a directory input"
),
)
parser.add_argument(
"--model",
default=DEFAULT_MODEL,
help=f"Ollama model name (default: {DEFAULT_MODEL})",
)
parser.add_argument(
"--endpoint",
default=DEFAULT_ENDPOINT,
help=f"Ollama generate endpoint (default: {DEFAULT_ENDPOINT})",
)
parser.add_argument(
"--timeout",
type=int,
default=1800,
help="HTTP timeout in seconds per chunk (default: 1800)",
)
parser.add_argument(
"--temperature",
type=float,
default=0.0,
help="Sampling temperature (default: 0.0)",
)
parser.add_argument(
"--num-predict",
type=int,
default=8192,
help="Maximum generated tokens per chunk (default: 8192)",
)
parser.add_argument(
"--num-ctx",
type=int,
default=32768,
help="Context window tokens per chunk (default: 32768)",
)
return parser.parse_args()
def build_prompt(source_name: str, transcript: str) -> str:
schema_text = json.dumps(OUTPUT_SCHEMA, ensure_ascii=False, indent=2)
return f"""{SYSTEM_INSTRUCTION}
Quelldatei:
{source_name}
Erwartete JSON-Struktur:
{schema_text}
Hinweise zur Ausgabe:
- Alle obersten Schlüssel müssen vorhanden sein.
- Verwende leere Listen, wenn keine Einträge vorhanden sind.
- Verwende null, wenn Verantwortliche, Sprecher oder Termine nicht erkennbar sind.
- "evidence" muss sich eng am Transkript orientieren.
- Ersetze technische Aussagen niemals durch eine vermeintlich korrektere Erklärung.
- Confidence-Werte sind ausdrücklich nicht erwünscht.
TRANSKRIPT:
--- BEGINN TRANSKRIPT ---
{transcript}
--- ENDE TRANSKRIPT ---
"""
def call_ollama(
endpoint: str,
model: str,
prompt: str,
timeout: int,
temperature: float,
num_predict: int | None,
num_ctx: int | None,
) -> tuple[str, dict[str, Any]]:
options: dict[str, Any] = {
"temperature": temperature,
}
if num_predict is not None:
options["num_predict"] = num_predict
if num_ctx is not None:
options["num_ctx"] = num_ctx
payload = {
"model": model,
"prompt": prompt,
"think": False,
"stream": False,
"format": "json",
"options": options,
}
started = time.perf_counter()
response = requests.post(endpoint, json=payload, timeout=timeout)
elapsed = time.perf_counter() - started
response.raise_for_status()
data = response.json()
text = response_text_from_ollama_data(data)
if not isinstance(text, str) or not text.strip():
raise ValueError("Ollama returned no usable response text.")
metadata = {
"model": data.get("model", model),
"elapsed_seconds": round(elapsed, 3),
"total_duration_ns": data.get("total_duration"),
"load_duration_ns": data.get("load_duration"),
"prompt_eval_count": data.get("prompt_eval_count"),
"prompt_eval_duration_ns": data.get("prompt_eval_duration"),
"eval_count": data.get("eval_count"),
"eval_duration_ns": data.get("eval_duration"),
}
return text.strip(), metadata
def response_text_from_ollama_data(data: dict[str, Any]) -> str | None:
text = data.get("response")
if isinstance(text, str) and text.strip():
return text
message = data.get("message")
if isinstance(message, dict):
content = message.get("content")
if isinstance(content, str) and content.strip():
return content
thinking = data.get("thinking")
if isinstance(thinking, str) and thinking.strip():
return thinking
return text if isinstance(text, str) else None
def iter_json_object_candidates(text: str) -> list[str]:
candidates: list[str] = []
start: int | None = None
depth = 0
in_string = False
escaped = False
for index, char in enumerate(text):
if in_string:
if escaped:
escaped = False
elif char == "\\":
escaped = True
elif char == '"':
in_string = False
continue
if char == '"':
in_string = True
continue
if char == "{":
if depth == 0:
start = index
depth += 1
continue
if char == "}" and depth:
depth -= 1
if depth == 0 and start is not None:
candidates.append(text[start : index + 1])
start = None
return candidates
def parse_json_response(text: str) -> dict[str, Any]:
try:
parsed = json.loads(text)
except json.JSONDecodeError:
parsed = None
for candidate in iter_json_object_candidates(text):
try:
candidate_json = json.loads(candidate)
except json.JSONDecodeError:
continue
if isinstance(candidate_json, dict):
parsed = candidate_json
if parsed is None:
raise
if not isinstance(parsed, dict):
raise ValueError("The model response is valid JSON but not a JSON object.")
return parsed
def text_from_value(value: Any, preferred_keys: tuple[str, ...]) -> str:
if isinstance(value, str):
return value.strip()
if isinstance(value, dict):
parts: list[str] = []
for key in preferred_keys:
item = value.get(key)
if item is None:
continue
text = str(item).strip()
if text:
parts.append(text)
if parts:
return " | ".join(parts)
return json.dumps(value, ensure_ascii=False, sort_keys=True)
return str(value).strip()
def normalize_current_schema(result: dict[str, Any]) -> dict[str, list[str]]:
return {
"facts": [
text
for item in result.get("facts", [])
if (
text := text_from_value(
item,
("speaker", "statement", "status", "evidence"),
)
)
],
"decisions": [
text
for item in result.get("decisions", [])
if (
text := text_from_value(
item,
("decision", "evidence"),
)
)
],
"todos": [
text
for item in result.get("todos", [])
if (
text := text_from_value(
item,
("task", "responsible", "deadline", "evidence"),
)
)
],
"questions": [
text
for item in result.get("open_questions", result.get("questions", []))
if (
text := text_from_value(
item,
("question", "evidence"),
)
)
],
"positions": [
text
for item in result.get("positions", [])
if (
text := text_from_value(
item,
("speaker", "position", "statement", "evidence"),
)
)
],
"technical": [
text
for item in result.get("technical_details", result.get("technical", []))
if (
text := text_from_value(
item,
("subject", "statement", "status", "evidence"),
)
)
],
}
def extraction_path_for_chunk(chunk_path: Path, output_dir: Path | None = None) -> Path:
match = NORMALIZED_CHUNK_RE.match(chunk_path.name)
if not match:
raise ValueError(
f"Not a normalized chunk filename: {chunk_path.name}"
)
directory = output_dir or chunk_path.parent
return directory / f"{match.group(1)}_extraction.json"
def find_normalized_chunks(input_dir: Path) -> list[Path]:
return sorted(input_dir.glob("chunk_*_normalized.txt"))
def extract_chunk(
chunk_path: Path,
output_path: Path,
model: str,
endpoint: str,
timeout: int,
temperature: float,
num_predict: int | None,
num_ctx: int | None,
) -> dict[str, list[str]]:
transcript = chunk_path.read_text(encoding="utf-8-sig").strip()
if not transcript:
raise ValueError(f"The input file is empty: {chunk_path}")
prompt = build_prompt(chunk_path.name, transcript)
raw_text, _metadata = call_ollama(
endpoint=endpoint,
model=model,
prompt=prompt,
timeout=timeout,
temperature=temperature,
num_predict=num_predict,
num_ctx=num_ctx,
)
try:
parsed = parse_json_response(raw_text)
except (json.JSONDecodeError, ValueError) as exc:
raw_path = output_path.with_suffix(".raw.txt")
raw_path.write_text(raw_text + "\n", encoding="utf-8")
raise ValueError(
f"Model output was not valid JSON. Raw output saved to: {raw_path}"
) from exc
extraction = normalize_current_schema(parsed)
output_path.parent.mkdir(parents=True, exist_ok=True)
output_path.write_text(
json.dumps(extraction, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
return extraction
def extract_input(
input_path: Path,
output: Path | None,
model: str,
endpoint: str,
timeout: int,
temperature: float,
num_predict: int | None,
num_ctx: int | None,
) -> list[Path]:
if input_path.is_file():
output_path = output or extraction_path_for_chunk(input_path)
extract_chunk(
input_path,
output_path,
model,
endpoint,
timeout,
temperature,
num_predict,
num_ctx,
)
return [output_path]
if not input_path.is_dir():
raise FileNotFoundError(f"Input path not found: {input_path}")
chunk_paths = find_normalized_chunks(input_path)
if not chunk_paths:
raise ValueError(f"No normalized chunks found in {input_path}")
output_dir = output if output is not None else input_path
output_paths: list[Path] = []
for chunk_path in chunk_paths:
output_path = extraction_path_for_chunk(chunk_path, output_dir)
print(
f"Extracting {chunk_path.name} -> {output_path.name}",
flush=True,
)
extract_chunk(
chunk_path,
output_path,
model,
endpoint,
timeout,
temperature,
num_predict,
num_ctx,
)
output_paths.append(output_path)
return output_paths
def main() -> int:
args = parse_args()
try:
output_paths = extract_input(
input_path=args.input,
output=args.output,
model=args.model,
endpoint=args.endpoint,
timeout=args.timeout,
temperature=args.temperature,
num_predict=args.num_predict,
num_ctx=args.num_ctx,
)
except requests.ConnectionError:
print(
"Error: Ollama is not reachable. Is `ollama serve` running?",
file=sys.stderr,
)
return 1
except requests.Timeout:
print("Error: The Ollama request timed out.", file=sys.stderr)
return 1
except requests.HTTPError as exc:
print(f"Error: Ollama returned an HTTP error: {exc}", file=sys.stderr)
return 1
except (OSError, UnicodeError, ValueError) as exc:
print(f"Error: {exc}", file=sys.stderr)
return 1
print(f"Input: {args.input}")
print(f"Processed chunks: {len(output_paths)}")
print(f"Extraction JSON files: {len(output_paths)}")
for output_path in output_paths:
print(output_path)
return 0
if __name__ == "__main__":
raise SystemExit(main())
@@ -0,0 +1,251 @@
#!/usr/bin/env python3
"""
Conservatively normalize a meeting transcript without rewriting its meaning.
The normalizer only performs low-risk cleanup:
- removes isolated filler sounds such as "äh" and "ähm"
- collapses immediate duplicate words or short duplicate phrases
- normalizes whitespace
- records every changed block in a JSON change log
It deliberately does NOT remove modal words, qualifiers, negations,
dates, numbers, responsibilities, technical statements, or commitments.
Examples:
python normalize_transcript.py meeting_chunks/chunk_01.txt
python normalize_transcript.py meeting_chunks/chunk_01.txt \
--output meeting_chunks/chunk_01_normalized.txt \
--changes meeting_chunks/chunk_01_changes.json
"""
from __future__ import annotations
import argparse
import json
import re
import sys
from dataclasses import asdict, dataclass
from pathlib import Path
# Only clearly non-semantic filler sounds.
FILLER_PATTERN = re.compile(
r"(?i)(?<![\wäöüß])(?:ähm+|äh+|hm+|mhm+)(?![\wäöüß])"
)
# Immediate duplicate single word:
# "die die Anlage" -> "die Anlage"
DUPLICATE_WORD_PATTERN = re.compile(
r"(?i)\b([A-Za-zÄÖÜäöüß][\wÄÖÜäöüß'-]*)"
r"(?:[\s,;:]+)\1\b"
)
# Immediate duplicate short phrase of two to four words:
# "die Lüftung, die Lüftung" -> "die Lüftung"
DUPLICATE_PHRASE_PATTERN = re.compile(
r"(?i)\b("
r"[A-Za-zÄÖÜäöüß][\wÄÖÜäöüß'-]*"
r"(?:\s+[A-Za-zÄÖÜäöüß][\wÄÖÜäöüß'-]*){1,3}"
r")"
r"(?:\s*[,;:]\s*|\s+)\1\b"
)
MULTISPACE_PATTERN = re.compile(r"[ \t]{2,}")
SPACE_BEFORE_PUNCT_PATTERN = re.compile(r"\s+([,.;:!?])")
MULTI_BLANK_PATTERN = re.compile(r"\n{3,}")
@dataclass(frozen=True)
class Change:
block_number: int
original: str
normalized: str
removed_fillers: list[str]
duplicate_reductions: list[str]
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Conservatively normalize a meeting transcript."
)
parser.add_argument("input_file", type=Path, help="Transcript text file")
parser.add_argument(
"-o",
"--output",
type=Path,
help="Normalized output file; default: <input>_normalized.txt",
)
parser.add_argument(
"--changes",
type=Path,
help="Change log JSON; default: <input>_changes.json",
)
return parser.parse_args()
def split_blocks(text: str) -> list[str]:
"""
Preserve transcript structure by treating blank-line-separated paragraphs
as atomic blocks. If there are no blank lines, use individual non-empty
lines as blocks.
"""
normalized_newlines = text.replace("\r\n", "\n").replace("\r", "\n").strip()
paragraphs = [
part.strip()
for part in re.split(r"\n\s*\n+", normalized_newlines)
if part.strip()
]
if len(paragraphs) > 1:
return paragraphs
return [line.strip() for line in normalized_newlines.splitlines() if line.strip()]
def remove_fillers(text: str) -> tuple[str, list[str]]:
removed = [match.group(0) for match in FILLER_PATTERN.finditer(text)]
cleaned = FILLER_PATTERN.sub("", text)
return cleaned, removed
def reduce_duplicate_phrases(text: str) -> tuple[str, list[str]]:
reductions: list[str] = []
current = text
# Repeat until stable because one correction may expose another.
for _ in range(5):
changed = False
def phrase_replacer(match: re.Match[str]) -> str:
nonlocal changed
changed = True
reductions.append(match.group(0))
return match.group(1)
updated = DUPLICATE_PHRASE_PATTERN.sub(phrase_replacer, current)
current = updated
def word_replacer(match: re.Match[str]) -> str:
nonlocal changed
changed = True
reductions.append(match.group(0))
return match.group(1)
updated = DUPLICATE_WORD_PATTERN.sub(word_replacer, current)
current = updated
if not changed:
break
return current, reductions
def tidy_spacing(text: str) -> str:
text = MULTISPACE_PATTERN.sub(" ", text)
text = SPACE_BEFORE_PUNCT_PATTERN.sub(r"\1", text)
text = re.sub(r"([,;:])([^\s\n])", r"\1 \2", text)
text = re.sub(r"\(\s+", "(", text)
text = re.sub(r"\s+\)", ")", text)
return text.strip(" \t,;:")
def normalize_block(block: str, block_number: int) -> tuple[str, Change | None]:
original = block
result, removed_fillers = remove_fillers(original)
result, duplicate_reductions = reduce_duplicate_phrases(result)
result = tidy_spacing(result)
if result == original:
return result, None
return result, Change(
block_number=block_number,
original=original,
normalized=result,
removed_fillers=removed_fillers,
duplicate_reductions=duplicate_reductions,
)
def main() -> int:
args = parse_args()
try:
if not args.input_file.is_file():
raise FileNotFoundError(f"Input file not found: {args.input_file}")
source = args.input_file.read_text(encoding="utf-8-sig")
if not source.strip():
raise ValueError("The input file is empty.")
output_path = args.output or args.input_file.with_name(
f"{args.input_file.stem}_normalized.txt"
)
changes_path = args.changes or args.input_file.with_name(
f"{args.input_file.stem}_changes.json"
)
blocks = split_blocks(source)
normalized_blocks: list[str] = []
changes: list[Change] = []
for block_number, block in enumerate(blocks, start=1):
normalized, change = normalize_block(block, block_number)
normalized_blocks.append(normalized)
if change is not None:
changes.append(change)
normalized_text = "\n\n".join(normalized_blocks).strip() + "\n"
output_path.parent.mkdir(parents=True, exist_ok=True)
changes_path.parent.mkdir(parents=True, exist_ok=True)
output_path.write_text(normalized_text, encoding="utf-8")
report = {
"source_file": args.input_file.name,
"output_file": output_path.name,
"blocks_total": len(blocks),
"blocks_changed": len(changes),
"policy": {
"removed": [
"isolated filler sounds such as äh, ähm, hm, mhm",
"immediate duplicate words",
"immediate duplicate phrases of two to four words",
"redundant whitespace",
],
"explicitly_preserved": [
"negations",
"modal words and qualifiers",
"dates, times, numbers and quantities",
"technical statements",
"responsibilities",
"deadlines",
"decisions and commitments",
],
"principle": "When uncertain, leave the text unchanged.",
},
"changes": [asdict(change) for change in changes],
}
changes_path.write_text(
json.dumps(report, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
print(f"Source: {args.input_file}")
print(f"Normalized: {output_path}")
print(f"Change log: {changes_path}")
print(f"Blocks total: {len(blocks)}")
print(f"Blocks changed: {len(changes)}")
return 0
except (OSError, UnicodeError, ValueError) as exc:
print(f"Error: {exc}", file=sys.stderr)
return 1
if __name__ == "__main__":
raise SystemExit(main())
+223
View File
@@ -0,0 +1,223 @@
#!/usr/bin/env python3
"""
Build a readable Markdown protocol from chunk extraction JSON files.
"""
from __future__ import annotations
import argparse
import json
import re
import sys
from pathlib import Path
from typing import Any
EXTRACTION_CATEGORIES = (
"facts",
"decisions",
"todos",
"questions",
"positions",
"technical",
)
CATEGORY_TITLES = {
"facts": "Facts",
"decisions": "Decisions",
"todos": "Action Items",
"questions": "Open Questions",
"positions": "Positions",
"technical": "Technical",
}
CHUNK_NUMBER_RE = re.compile(r"chunk_(\d+)_")
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Build a Markdown meeting protocol from extraction JSON files."
)
parser.add_argument(
"input_dir",
type=Path,
help="Directory containing chunk_XX_extraction.json files",
)
parser.add_argument(
"-o",
"--output",
type=Path,
help="Output Markdown file; default: <input-dir>/meeting_protocol.md",
)
return parser.parse_args()
def chunk_sort_key(path: Path) -> tuple[int, str]:
match = CHUNK_NUMBER_RE.search(path.name)
if not match:
return (sys.maxsize, path.name)
return (int(match.group(1)), path.name)
def load_json(path: Path) -> Any:
return json.loads(path.read_text(encoding="utf-8-sig"))
def find_extraction_files(input_dir: Path) -> list[Path]:
return sorted(
input_dir.glob("chunk_*_extraction.json"),
key=chunk_sort_key,
)
def normalize_items(value: Any) -> list[str]:
if not isinstance(value, list):
return []
items: list[str] = []
for item in value:
if isinstance(item, str):
text = item.strip()
else:
text = json.dumps(item, ensure_ascii=False, sort_keys=True)
if text:
items.append(text)
return items
def collect_extractions(
extraction_files: list[Path],
) -> dict[str, list[str]]:
collected = {category: [] for category in EXTRACTION_CATEGORIES}
for path in extraction_files:
data = load_json(path)
if not isinstance(data, dict):
raise ValueError(
f"Extraction file must contain a JSON object: {path}"
)
for category in EXTRACTION_CATEGORIES:
collected[category].extend(
normalize_items(data.get(category, []))
)
return collected
def find_topic_files(input_dir: Path) -> list[Path]:
return sorted(
input_dir.glob("chunk_*_normalized_windowed_segments.json"),
key=chunk_sort_key,
)
def collect_topics(input_dir: Path) -> list[str]:
topics: list[str] = []
for path in find_topic_files(input_dir):
data = load_json(path)
if not isinstance(data, dict):
continue
source = data.get("source", {})
source_file = (
source.get("file")
if isinstance(source, dict)
else None
)
label = source_file or path.name
segments = data.get("segments", [])
if not isinstance(segments, list):
continue
for segment in segments:
if not isinstance(segment, dict):
continue
segment_id = segment.get("segment_id", "segment")
start_block = segment.get("start_block", "?")
end_block = segment.get("end_block", "?")
topics.append(
f"{label}: {segment_id} (blocks {start_block}-{end_block})"
)
return topics
def render_list(items: list[str]) -> list[str]:
if not items:
return ["- None recorded."]
return [f"- {item}" for item in items]
def build_protocol_markdown(
input_dir: Path,
extraction_files: list[Path],
) -> str:
extractions = collect_extractions(extraction_files)
topics = collect_topics(input_dir)
lines = [
"# Meeting Protocol",
"",
"## Topics",
*render_list(topics),
"",
"## Facts",
*render_list(extractions["facts"]),
"",
"## Decisions",
*render_list(extractions["decisions"]),
"",
"## Open Questions",
*render_list(extractions["questions"]),
"",
"## Action Items",
*render_list(extractions["todos"]),
"",
"## Positions",
*render_list(extractions["positions"]),
"",
"## Technical",
*render_list(extractions["technical"]),
"",
]
return "\n".join(lines)
def build_protocol(input_dir: Path, output_path: Path | None = None) -> Path:
if not input_dir.is_dir():
raise FileNotFoundError(f"Input directory not found: {input_dir}")
extraction_files = find_extraction_files(input_dir)
if not extraction_files:
raise ValueError(
f"No extraction JSON files found in {input_dir}"
)
output = output_path or input_dir / "meeting_protocol.md"
markdown = build_protocol_markdown(input_dir, extraction_files)
output.write_text(markdown, encoding="utf-8")
return output
def main() -> int:
args = parse_args()
try:
extraction_files = find_extraction_files(args.input_dir)
output_path = build_protocol(args.input_dir, args.output)
except (OSError, UnicodeError, ValueError, json.JSONDecodeError) as exc:
print(f"Error: {exc}", file=sys.stderr)
return 1
print(f"Input: {args.input_dir}")
print(f"Extraction JSON files: {len(extraction_files)}")
print(f"Protocol: {output_path}")
return 0
if __name__ == "__main__":
raise SystemExit(main())
+124
View File
@@ -0,0 +1,124 @@
import json
import tempfile
import unittest
from pathlib import Path
from src.meeting_lab.extraction.extract_chunks import (
EXTRACTION_CATEGORIES,
extraction_path_for_chunk,
normalize_current_schema,
parse_json_response,
)
from src.meeting_lab.protocol.build_protocol import build_protocol
class ExtractionProtocolTests(unittest.TestCase):
def test_extraction_path_drops_normalized_suffix(self) -> None:
path = Path("chunks/chunk_01_normalized.txt")
self.assertEqual(
extraction_path_for_chunk(path),
Path("chunks/chunk_01_extraction.json"),
)
def test_normalize_current_schema_maps_legacy_response(self) -> None:
result = normalize_current_schema(
{
"facts": [
{
"speaker": "A",
"statement": "Fact one",
"status": "clear",
"evidence": "Fact",
}
],
"decisions": [
{
"decision": "Decision one",
"evidence": "Decision",
}
],
"todos": [
{
"task": "Todo one",
"responsible": "B",
"deadline": None,
"evidence": "Todo",
}
],
"open_questions": [
{
"question": "Question one",
"evidence": "Question",
}
],
"technical_details": [
{
"subject": "System",
"statement": "Technical one",
"status": "clear",
"evidence": "Technical",
}
],
}
)
self.assertEqual(set(result), set(EXTRACTION_CATEGORIES))
self.assertEqual(result["positions"], [])
self.assertIn("Fact one", result["facts"][0])
self.assertIn("Decision one", result["decisions"][0])
self.assertIn("Todo one", result["todos"][0])
self.assertIn("Question one", result["questions"][0])
self.assertIn("Technical one", result["technical"][0])
def test_parse_json_response_uses_final_object_after_thinking(self) -> None:
text = """
Thinking:
I will reason about the transcript first.
{"facts": ["draft"], "decisions": []}
Final answer:
{
"facts": ["final"],
"decisions": [],
"todos": [],
"questions": [],
"positions": [],
"technical": []
}
"""
parsed = parse_json_response(text)
self.assertEqual(parsed["facts"], ["final"])
self.assertEqual(set(parsed), set(EXTRACTION_CATEGORIES))
def test_build_protocol_groups_extraction_items(self) -> None:
with tempfile.TemporaryDirectory() as directory:
input_dir = Path(directory)
(input_dir / "chunk_01_extraction.json").write_text(
json.dumps(
{
"facts": ["Fact one"],
"decisions": ["Decision one"],
"todos": ["Todo one"],
"questions": ["Question one"],
"positions": [],
"technical": [],
}
),
encoding="utf-8",
)
output_path = build_protocol(input_dir)
markdown = output_path.read_text(encoding="utf-8")
self.assertIn("# Meeting Protocol", markdown)
self.assertIn("## Facts\n- Fact one", markdown)
self.assertIn("## Decisions\n- Decision one", markdown)
self.assertIn("## Open Questions\n- Question one", markdown)
self.assertIn("## Action Items\n- Todo one", markdown)
if __name__ == "__main__":
unittest.main()