Implement first end-to-end meeting analysis pipeline
This commit is contained in:
@@ -32,6 +32,13 @@ htmlcov/
|
||||
experiments/**/output/
|
||||
experiments/**/results/
|
||||
|
||||
# Pipeline runtime artifacts
|
||||
samples/raw/
|
||||
samples/whisper/
|
||||
**/chunk_*_extraction.json
|
||||
**/meeting_protocol.md
|
||||
**/*.raw.txt
|
||||
|
||||
# Lokale Meetings (niemals versionieren)
|
||||
meeting_data/
|
||||
recordings/
|
||||
|
||||
@@ -0,0 +1,139 @@
|
||||
import argparse
|
||||
import json
|
||||
import math
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
def is_nonfinite_number(value: Any) -> bool:
|
||||
"""Prüft auf NaN sowie positive oder negative Unendlichkeit."""
|
||||
return isinstance(value, float) and not math.isfinite(value)
|
||||
|
||||
|
||||
def sanitize_nonfinite_values(value: Any) -> Any:
|
||||
"""
|
||||
Ersetzt NaN und Infinity rekursiv durch None.
|
||||
|
||||
None wird in JSON als null geschrieben.
|
||||
"""
|
||||
if is_nonfinite_number(value):
|
||||
return None
|
||||
|
||||
if isinstance(value, dict):
|
||||
return {
|
||||
key: sanitize_nonfinite_values(item)
|
||||
for key, item in value.items()
|
||||
}
|
||||
|
||||
if isinstance(value, list):
|
||||
return [
|
||||
sanitize_nonfinite_values(item)
|
||||
for item in value
|
||||
]
|
||||
|
||||
return value
|
||||
|
||||
|
||||
def is_invalid_empty_segment(segment: dict[str, Any]) -> bool:
|
||||
"""Erkennt technisch leere Whisper-Artefakte."""
|
||||
|
||||
text = str(segment.get("text", "")).strip()
|
||||
start = segment.get("start")
|
||||
end = segment.get("end")
|
||||
avg_logprob = segment.get("avg_logprob")
|
||||
|
||||
logprob_is_nan = (
|
||||
isinstance(avg_logprob, float)
|
||||
and math.isnan(avg_logprob)
|
||||
)
|
||||
|
||||
return (
|
||||
not text
|
||||
and start == end
|
||||
and logprob_is_nan
|
||||
)
|
||||
|
||||
|
||||
def clean_whisper_json(
|
||||
input_path: Path,
|
||||
output_path: Path,
|
||||
) -> None:
|
||||
with input_path.open("r", encoding="utf-8") as file:
|
||||
data = json.load(file)
|
||||
|
||||
original_segments = data.get("segments", [])
|
||||
|
||||
cleaned_segments = [
|
||||
segment
|
||||
for segment in original_segments
|
||||
if not is_invalid_empty_segment(segment)
|
||||
]
|
||||
|
||||
for new_id, segment in enumerate(cleaned_segments):
|
||||
segment["id"] = new_id
|
||||
|
||||
data["segments"] = cleaned_segments
|
||||
|
||||
data["text"] = " ".join(
|
||||
str(segment.get("text", "")).strip()
|
||||
for segment in cleaned_segments
|
||||
if str(segment.get("text", "")).strip()
|
||||
)
|
||||
|
||||
# Alle noch vorhandenen NaN-/Infinity-Werte durch null ersetzen.
|
||||
data = sanitize_nonfinite_values(data)
|
||||
|
||||
# Erst vollständig in einen String serialisieren.
|
||||
# Dadurch bleibt keine unvollständige Ausgabedatei zurück,
|
||||
# falls doch noch ein Fehler auftritt.
|
||||
json_content = json.dumps(
|
||||
data,
|
||||
ensure_ascii=False,
|
||||
indent=2,
|
||||
allow_nan=False,
|
||||
)
|
||||
|
||||
output_path.write_text(
|
||||
json_content + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
removed = len(original_segments) - len(cleaned_segments)
|
||||
|
||||
print(f"Eingabedatei: {input_path}")
|
||||
print(f"Ausgabedatei: {output_path}")
|
||||
print(f"Segmente vorher: {len(original_segments)}")
|
||||
print(f"Segmente nachher: {len(cleaned_segments)}")
|
||||
print(f"Segmente entfernt: {removed}")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(
|
||||
description=(
|
||||
"Entfernt technisch leere Artefakte aus "
|
||||
"einer MLX-Whisper-JSON-Datei."
|
||||
)
|
||||
)
|
||||
parser.add_argument(
|
||||
"input",
|
||||
type=Path,
|
||||
help="Whisper-JSON-Datei",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-o",
|
||||
"--output",
|
||||
type=Path,
|
||||
help="Ausgabedatei",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
output_path = args.output or args.input.with_name(
|
||||
f"{args.input.stem}_cleaned.json"
|
||||
)
|
||||
|
||||
clean_whisper_json(args.input, output_path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,550 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Extract structured meeting information from normalized transcript chunks.
|
||||
|
||||
This module is adapted from /Users/tazl/quick-whisper-test/summarize.py.
|
||||
It keeps the same Ollama extraction flow and conservative prompt rules, but
|
||||
writes the current meeting-lab extraction schema.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
DEFAULT_MODEL = "qwen3:8b"
|
||||
DEFAULT_ENDPOINT = "http://localhost:11434/api/generate"
|
||||
|
||||
EXTRACTION_CATEGORIES = (
|
||||
"facts",
|
||||
"decisions",
|
||||
"todos",
|
||||
"questions",
|
||||
"positions",
|
||||
"technical",
|
||||
)
|
||||
|
||||
NORMALIZED_CHUNK_RE = re.compile(r"^(chunk_\d+)_normalized\.txt$")
|
||||
|
||||
|
||||
SYSTEM_INSTRUCTION = """
|
||||
Du extrahierst Informationen aus Meeting-Transkripten.
|
||||
|
||||
Arbeite ausschließlich mit dem vorgelegten Transkript.
|
||||
Verwende kein eigenes Fachwissen, keine Vermutungen und keine üblichen
|
||||
Funktionsweisen technischer Systeme.
|
||||
|
||||
Regeln:
|
||||
1. Erfinde nichts.
|
||||
2. Interpretiere technische Aussagen nicht über den Wortlaut hinaus.
|
||||
3. Korrigiere keine Aussagen anhand vermeintlichen Weltwissens.
|
||||
4. Wenn etwas widersprüchlich oder unklar ist, kennzeichne es als unklar.
|
||||
5. Übernimm wichtige technische Aussagen möglichst nah am Wortlaut.
|
||||
6. Nenne bei Fakten nach Möglichkeit den Sprecher.
|
||||
7. Ein Beschluss ist nur dann ein Beschluss, wenn im Text eine Einigung,
|
||||
Freigabe oder verbindliche Festlegung erkennbar ist.
|
||||
8. Eine Aufgabe ist nur dann eine Aufgabe, wenn eine Handlung und möglichst
|
||||
eine verantwortliche Person oder Organisation erkennbar sind.
|
||||
9. Gib ausschließlich gültiges JSON aus. Kein Markdown, keine Erläuterungen.
|
||||
""".strip()
|
||||
|
||||
|
||||
OUTPUT_SCHEMA = {
|
||||
"chunk": {
|
||||
"source_file": "string",
|
||||
"summary": "Kurze, rein inhaltsbezogene Beschreibung des Chunks oder leer",
|
||||
},
|
||||
"participants": ["string"],
|
||||
"topics": ["string"],
|
||||
"facts": [
|
||||
{
|
||||
"speaker": "string oder null",
|
||||
"statement": "Aussage möglichst nah am Wortlaut",
|
||||
"status": "clear | unclear | contradictory",
|
||||
"evidence": "Kurzes wörtliches oder nahezu wörtliches Textfragment",
|
||||
}
|
||||
],
|
||||
"decisions": [
|
||||
{
|
||||
"decision": "string",
|
||||
"evidence": "Kurzes Textfragment",
|
||||
}
|
||||
],
|
||||
"todos": [
|
||||
{
|
||||
"task": "string",
|
||||
"responsible": "string oder null",
|
||||
"deadline": "string oder null",
|
||||
"evidence": "Kurzes Textfragment",
|
||||
}
|
||||
],
|
||||
"open_questions": [
|
||||
{
|
||||
"question": "string",
|
||||
"evidence": "Kurzes Textfragment",
|
||||
}
|
||||
],
|
||||
"technical_details": [
|
||||
{
|
||||
"subject": "string",
|
||||
"statement": "Technische Aussage möglichst nah am Wortlaut",
|
||||
"status": "clear | unclear | contradictory",
|
||||
"evidence": "Kurzes Textfragment",
|
||||
}
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Extract structured facts from normalized meeting chunks."
|
||||
)
|
||||
parser.add_argument(
|
||||
"input",
|
||||
type=Path,
|
||||
help=(
|
||||
"A chunk_XX_normalized.txt file or a directory containing "
|
||||
"normalized chunk files"
|
||||
),
|
||||
)
|
||||
parser.add_argument(
|
||||
"-o",
|
||||
"--output",
|
||||
type=Path,
|
||||
help=(
|
||||
"Output JSON path for a single input file, or output directory "
|
||||
"for a directory input"
|
||||
),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--model",
|
||||
default=DEFAULT_MODEL,
|
||||
help=f"Ollama model name (default: {DEFAULT_MODEL})",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--endpoint",
|
||||
default=DEFAULT_ENDPOINT,
|
||||
help=f"Ollama generate endpoint (default: {DEFAULT_ENDPOINT})",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--timeout",
|
||||
type=int,
|
||||
default=1800,
|
||||
help="HTTP timeout in seconds per chunk (default: 1800)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--temperature",
|
||||
type=float,
|
||||
default=0.0,
|
||||
help="Sampling temperature (default: 0.0)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--num-predict",
|
||||
type=int,
|
||||
default=8192,
|
||||
help="Maximum generated tokens per chunk (default: 8192)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--num-ctx",
|
||||
type=int,
|
||||
default=32768,
|
||||
help="Context window tokens per chunk (default: 32768)",
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def build_prompt(source_name: str, transcript: str) -> str:
|
||||
schema_text = json.dumps(OUTPUT_SCHEMA, ensure_ascii=False, indent=2)
|
||||
|
||||
return f"""{SYSTEM_INSTRUCTION}
|
||||
|
||||
Quelldatei:
|
||||
{source_name}
|
||||
|
||||
Erwartete JSON-Struktur:
|
||||
{schema_text}
|
||||
|
||||
Hinweise zur Ausgabe:
|
||||
- Alle obersten Schlüssel müssen vorhanden sein.
|
||||
- Verwende leere Listen, wenn keine Einträge vorhanden sind.
|
||||
- Verwende null, wenn Verantwortliche, Sprecher oder Termine nicht erkennbar sind.
|
||||
- "evidence" muss sich eng am Transkript orientieren.
|
||||
- Ersetze technische Aussagen niemals durch eine vermeintlich korrektere Erklärung.
|
||||
- Confidence-Werte sind ausdrücklich nicht erwünscht.
|
||||
|
||||
TRANSKRIPT:
|
||||
--- BEGINN TRANSKRIPT ---
|
||||
{transcript}
|
||||
--- ENDE TRANSKRIPT ---
|
||||
"""
|
||||
|
||||
|
||||
def call_ollama(
|
||||
endpoint: str,
|
||||
model: str,
|
||||
prompt: str,
|
||||
timeout: int,
|
||||
temperature: float,
|
||||
num_predict: int | None,
|
||||
num_ctx: int | None,
|
||||
) -> tuple[str, dict[str, Any]]:
|
||||
options: dict[str, Any] = {
|
||||
"temperature": temperature,
|
||||
}
|
||||
if num_predict is not None:
|
||||
options["num_predict"] = num_predict
|
||||
if num_ctx is not None:
|
||||
options["num_ctx"] = num_ctx
|
||||
|
||||
payload = {
|
||||
"model": model,
|
||||
"prompt": prompt,
|
||||
"think": False,
|
||||
"stream": False,
|
||||
"format": "json",
|
||||
"options": options,
|
||||
}
|
||||
|
||||
started = time.perf_counter()
|
||||
response = requests.post(endpoint, json=payload, timeout=timeout)
|
||||
elapsed = time.perf_counter() - started
|
||||
response.raise_for_status()
|
||||
|
||||
data = response.json()
|
||||
text = response_text_from_ollama_data(data)
|
||||
if not isinstance(text, str) or not text.strip():
|
||||
raise ValueError("Ollama returned no usable response text.")
|
||||
|
||||
metadata = {
|
||||
"model": data.get("model", model),
|
||||
"elapsed_seconds": round(elapsed, 3),
|
||||
"total_duration_ns": data.get("total_duration"),
|
||||
"load_duration_ns": data.get("load_duration"),
|
||||
"prompt_eval_count": data.get("prompt_eval_count"),
|
||||
"prompt_eval_duration_ns": data.get("prompt_eval_duration"),
|
||||
"eval_count": data.get("eval_count"),
|
||||
"eval_duration_ns": data.get("eval_duration"),
|
||||
}
|
||||
return text.strip(), metadata
|
||||
|
||||
|
||||
def response_text_from_ollama_data(data: dict[str, Any]) -> str | None:
|
||||
text = data.get("response")
|
||||
if isinstance(text, str) and text.strip():
|
||||
return text
|
||||
|
||||
message = data.get("message")
|
||||
if isinstance(message, dict):
|
||||
content = message.get("content")
|
||||
if isinstance(content, str) and content.strip():
|
||||
return content
|
||||
|
||||
thinking = data.get("thinking")
|
||||
if isinstance(thinking, str) and thinking.strip():
|
||||
return thinking
|
||||
|
||||
return text if isinstance(text, str) else None
|
||||
|
||||
|
||||
def iter_json_object_candidates(text: str) -> list[str]:
|
||||
candidates: list[str] = []
|
||||
start: int | None = None
|
||||
depth = 0
|
||||
in_string = False
|
||||
escaped = False
|
||||
|
||||
for index, char in enumerate(text):
|
||||
if in_string:
|
||||
if escaped:
|
||||
escaped = False
|
||||
elif char == "\\":
|
||||
escaped = True
|
||||
elif char == '"':
|
||||
in_string = False
|
||||
continue
|
||||
|
||||
if char == '"':
|
||||
in_string = True
|
||||
continue
|
||||
|
||||
if char == "{":
|
||||
if depth == 0:
|
||||
start = index
|
||||
depth += 1
|
||||
continue
|
||||
|
||||
if char == "}" and depth:
|
||||
depth -= 1
|
||||
if depth == 0 and start is not None:
|
||||
candidates.append(text[start : index + 1])
|
||||
start = None
|
||||
|
||||
return candidates
|
||||
|
||||
|
||||
def parse_json_response(text: str) -> dict[str, Any]:
|
||||
try:
|
||||
parsed = json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
parsed = None
|
||||
for candidate in iter_json_object_candidates(text):
|
||||
try:
|
||||
candidate_json = json.loads(candidate)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if isinstance(candidate_json, dict):
|
||||
parsed = candidate_json
|
||||
if parsed is None:
|
||||
raise
|
||||
|
||||
if not isinstance(parsed, dict):
|
||||
raise ValueError("The model response is valid JSON but not a JSON object.")
|
||||
return parsed
|
||||
|
||||
|
||||
def text_from_value(value: Any, preferred_keys: tuple[str, ...]) -> str:
|
||||
if isinstance(value, str):
|
||||
return value.strip()
|
||||
|
||||
if isinstance(value, dict):
|
||||
parts: list[str] = []
|
||||
for key in preferred_keys:
|
||||
item = value.get(key)
|
||||
if item is None:
|
||||
continue
|
||||
text = str(item).strip()
|
||||
if text:
|
||||
parts.append(text)
|
||||
if parts:
|
||||
return " | ".join(parts)
|
||||
return json.dumps(value, ensure_ascii=False, sort_keys=True)
|
||||
|
||||
return str(value).strip()
|
||||
|
||||
|
||||
def normalize_current_schema(result: dict[str, Any]) -> dict[str, list[str]]:
|
||||
return {
|
||||
"facts": [
|
||||
text
|
||||
for item in result.get("facts", [])
|
||||
if (
|
||||
text := text_from_value(
|
||||
item,
|
||||
("speaker", "statement", "status", "evidence"),
|
||||
)
|
||||
)
|
||||
],
|
||||
"decisions": [
|
||||
text
|
||||
for item in result.get("decisions", [])
|
||||
if (
|
||||
text := text_from_value(
|
||||
item,
|
||||
("decision", "evidence"),
|
||||
)
|
||||
)
|
||||
],
|
||||
"todos": [
|
||||
text
|
||||
for item in result.get("todos", [])
|
||||
if (
|
||||
text := text_from_value(
|
||||
item,
|
||||
("task", "responsible", "deadline", "evidence"),
|
||||
)
|
||||
)
|
||||
],
|
||||
"questions": [
|
||||
text
|
||||
for item in result.get("open_questions", result.get("questions", []))
|
||||
if (
|
||||
text := text_from_value(
|
||||
item,
|
||||
("question", "evidence"),
|
||||
)
|
||||
)
|
||||
],
|
||||
"positions": [
|
||||
text
|
||||
for item in result.get("positions", [])
|
||||
if (
|
||||
text := text_from_value(
|
||||
item,
|
||||
("speaker", "position", "statement", "evidence"),
|
||||
)
|
||||
)
|
||||
],
|
||||
"technical": [
|
||||
text
|
||||
for item in result.get("technical_details", result.get("technical", []))
|
||||
if (
|
||||
text := text_from_value(
|
||||
item,
|
||||
("subject", "statement", "status", "evidence"),
|
||||
)
|
||||
)
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def extraction_path_for_chunk(chunk_path: Path, output_dir: Path | None = None) -> Path:
|
||||
match = NORMALIZED_CHUNK_RE.match(chunk_path.name)
|
||||
if not match:
|
||||
raise ValueError(
|
||||
f"Not a normalized chunk filename: {chunk_path.name}"
|
||||
)
|
||||
directory = output_dir or chunk_path.parent
|
||||
return directory / f"{match.group(1)}_extraction.json"
|
||||
|
||||
|
||||
def find_normalized_chunks(input_dir: Path) -> list[Path]:
|
||||
return sorted(input_dir.glob("chunk_*_normalized.txt"))
|
||||
|
||||
|
||||
def extract_chunk(
|
||||
chunk_path: Path,
|
||||
output_path: Path,
|
||||
model: str,
|
||||
endpoint: str,
|
||||
timeout: int,
|
||||
temperature: float,
|
||||
num_predict: int | None,
|
||||
num_ctx: int | None,
|
||||
) -> dict[str, list[str]]:
|
||||
transcript = chunk_path.read_text(encoding="utf-8-sig").strip()
|
||||
if not transcript:
|
||||
raise ValueError(f"The input file is empty: {chunk_path}")
|
||||
|
||||
prompt = build_prompt(chunk_path.name, transcript)
|
||||
raw_text, _metadata = call_ollama(
|
||||
endpoint=endpoint,
|
||||
model=model,
|
||||
prompt=prompt,
|
||||
timeout=timeout,
|
||||
temperature=temperature,
|
||||
num_predict=num_predict,
|
||||
num_ctx=num_ctx,
|
||||
)
|
||||
|
||||
try:
|
||||
parsed = parse_json_response(raw_text)
|
||||
except (json.JSONDecodeError, ValueError) as exc:
|
||||
raw_path = output_path.with_suffix(".raw.txt")
|
||||
raw_path.write_text(raw_text + "\n", encoding="utf-8")
|
||||
raise ValueError(
|
||||
f"Model output was not valid JSON. Raw output saved to: {raw_path}"
|
||||
) from exc
|
||||
|
||||
extraction = normalize_current_schema(parsed)
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
output_path.write_text(
|
||||
json.dumps(extraction, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
return extraction
|
||||
|
||||
|
||||
def extract_input(
|
||||
input_path: Path,
|
||||
output: Path | None,
|
||||
model: str,
|
||||
endpoint: str,
|
||||
timeout: int,
|
||||
temperature: float,
|
||||
num_predict: int | None,
|
||||
num_ctx: int | None,
|
||||
) -> list[Path]:
|
||||
if input_path.is_file():
|
||||
output_path = output or extraction_path_for_chunk(input_path)
|
||||
extract_chunk(
|
||||
input_path,
|
||||
output_path,
|
||||
model,
|
||||
endpoint,
|
||||
timeout,
|
||||
temperature,
|
||||
num_predict,
|
||||
num_ctx,
|
||||
)
|
||||
return [output_path]
|
||||
|
||||
if not input_path.is_dir():
|
||||
raise FileNotFoundError(f"Input path not found: {input_path}")
|
||||
|
||||
chunk_paths = find_normalized_chunks(input_path)
|
||||
if not chunk_paths:
|
||||
raise ValueError(f"No normalized chunks found in {input_path}")
|
||||
|
||||
output_dir = output if output is not None else input_path
|
||||
output_paths: list[Path] = []
|
||||
|
||||
for chunk_path in chunk_paths:
|
||||
output_path = extraction_path_for_chunk(chunk_path, output_dir)
|
||||
print(
|
||||
f"Extracting {chunk_path.name} -> {output_path.name}",
|
||||
flush=True,
|
||||
)
|
||||
extract_chunk(
|
||||
chunk_path,
|
||||
output_path,
|
||||
model,
|
||||
endpoint,
|
||||
timeout,
|
||||
temperature,
|
||||
num_predict,
|
||||
num_ctx,
|
||||
)
|
||||
output_paths.append(output_path)
|
||||
|
||||
return output_paths
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
|
||||
try:
|
||||
output_paths = extract_input(
|
||||
input_path=args.input,
|
||||
output=args.output,
|
||||
model=args.model,
|
||||
endpoint=args.endpoint,
|
||||
timeout=args.timeout,
|
||||
temperature=args.temperature,
|
||||
num_predict=args.num_predict,
|
||||
num_ctx=args.num_ctx,
|
||||
)
|
||||
except requests.ConnectionError:
|
||||
print(
|
||||
"Error: Ollama is not reachable. Is `ollama serve` running?",
|
||||
file=sys.stderr,
|
||||
)
|
||||
return 1
|
||||
except requests.Timeout:
|
||||
print("Error: The Ollama request timed out.", file=sys.stderr)
|
||||
return 1
|
||||
except requests.HTTPError as exc:
|
||||
print(f"Error: Ollama returned an HTTP error: {exc}", file=sys.stderr)
|
||||
return 1
|
||||
except (OSError, UnicodeError, ValueError) as exc:
|
||||
print(f"Error: {exc}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
print(f"Input: {args.input}")
|
||||
print(f"Processed chunks: {len(output_paths)}")
|
||||
print(f"Extraction JSON files: {len(output_paths)}")
|
||||
for output_path in output_paths:
|
||||
print(output_path)
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,251 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Conservatively normalize a meeting transcript without rewriting its meaning.
|
||||
|
||||
The normalizer only performs low-risk cleanup:
|
||||
- removes isolated filler sounds such as "äh" and "ähm"
|
||||
- collapses immediate duplicate words or short duplicate phrases
|
||||
- normalizes whitespace
|
||||
- records every changed block in a JSON change log
|
||||
|
||||
It deliberately does NOT remove modal words, qualifiers, negations,
|
||||
dates, numbers, responsibilities, technical statements, or commitments.
|
||||
|
||||
Examples:
|
||||
python normalize_transcript.py meeting_chunks/chunk_01.txt
|
||||
|
||||
python normalize_transcript.py meeting_chunks/chunk_01.txt \
|
||||
--output meeting_chunks/chunk_01_normalized.txt \
|
||||
--changes meeting_chunks/chunk_01_changes.json
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
from dataclasses import asdict, dataclass
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
# Only clearly non-semantic filler sounds.
|
||||
FILLER_PATTERN = re.compile(
|
||||
r"(?i)(?<![\wäöüß])(?:ähm+|äh+|hm+|mhm+)(?![\wäöüß])"
|
||||
)
|
||||
|
||||
# Immediate duplicate single word:
|
||||
# "die die Anlage" -> "die Anlage"
|
||||
DUPLICATE_WORD_PATTERN = re.compile(
|
||||
r"(?i)\b([A-Za-zÄÖÜäöüß][\wÄÖÜäöüß'-]*)"
|
||||
r"(?:[\s,;:]+)\1\b"
|
||||
)
|
||||
|
||||
# Immediate duplicate short phrase of two to four words:
|
||||
# "die Lüftung, die Lüftung" -> "die Lüftung"
|
||||
DUPLICATE_PHRASE_PATTERN = re.compile(
|
||||
r"(?i)\b("
|
||||
r"[A-Za-zÄÖÜäöüß][\wÄÖÜäöüß'-]*"
|
||||
r"(?:\s+[A-Za-zÄÖÜäöüß][\wÄÖÜäöüß'-]*){1,3}"
|
||||
r")"
|
||||
r"(?:\s*[,;:]\s*|\s+)\1\b"
|
||||
)
|
||||
|
||||
MULTISPACE_PATTERN = re.compile(r"[ \t]{2,}")
|
||||
SPACE_BEFORE_PUNCT_PATTERN = re.compile(r"\s+([,.;:!?])")
|
||||
MULTI_BLANK_PATTERN = re.compile(r"\n{3,}")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Change:
|
||||
block_number: int
|
||||
original: str
|
||||
normalized: str
|
||||
removed_fillers: list[str]
|
||||
duplicate_reductions: list[str]
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Conservatively normalize a meeting transcript."
|
||||
)
|
||||
parser.add_argument("input_file", type=Path, help="Transcript text file")
|
||||
parser.add_argument(
|
||||
"-o",
|
||||
"--output",
|
||||
type=Path,
|
||||
help="Normalized output file; default: <input>_normalized.txt",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--changes",
|
||||
type=Path,
|
||||
help="Change log JSON; default: <input>_changes.json",
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def split_blocks(text: str) -> list[str]:
|
||||
"""
|
||||
Preserve transcript structure by treating blank-line-separated paragraphs
|
||||
as atomic blocks. If there are no blank lines, use individual non-empty
|
||||
lines as blocks.
|
||||
"""
|
||||
normalized_newlines = text.replace("\r\n", "\n").replace("\r", "\n").strip()
|
||||
|
||||
paragraphs = [
|
||||
part.strip()
|
||||
for part in re.split(r"\n\s*\n+", normalized_newlines)
|
||||
if part.strip()
|
||||
]
|
||||
if len(paragraphs) > 1:
|
||||
return paragraphs
|
||||
|
||||
return [line.strip() for line in normalized_newlines.splitlines() if line.strip()]
|
||||
|
||||
|
||||
def remove_fillers(text: str) -> tuple[str, list[str]]:
|
||||
removed = [match.group(0) for match in FILLER_PATTERN.finditer(text)]
|
||||
cleaned = FILLER_PATTERN.sub("", text)
|
||||
return cleaned, removed
|
||||
|
||||
|
||||
def reduce_duplicate_phrases(text: str) -> tuple[str, list[str]]:
|
||||
reductions: list[str] = []
|
||||
current = text
|
||||
|
||||
# Repeat until stable because one correction may expose another.
|
||||
for _ in range(5):
|
||||
changed = False
|
||||
|
||||
def phrase_replacer(match: re.Match[str]) -> str:
|
||||
nonlocal changed
|
||||
changed = True
|
||||
reductions.append(match.group(0))
|
||||
return match.group(1)
|
||||
|
||||
updated = DUPLICATE_PHRASE_PATTERN.sub(phrase_replacer, current)
|
||||
current = updated
|
||||
|
||||
def word_replacer(match: re.Match[str]) -> str:
|
||||
nonlocal changed
|
||||
changed = True
|
||||
reductions.append(match.group(0))
|
||||
return match.group(1)
|
||||
|
||||
updated = DUPLICATE_WORD_PATTERN.sub(word_replacer, current)
|
||||
current = updated
|
||||
|
||||
if not changed:
|
||||
break
|
||||
|
||||
return current, reductions
|
||||
|
||||
|
||||
def tidy_spacing(text: str) -> str:
|
||||
text = MULTISPACE_PATTERN.sub(" ", text)
|
||||
text = SPACE_BEFORE_PUNCT_PATTERN.sub(r"\1", text)
|
||||
text = re.sub(r"([,;:])([^\s\n])", r"\1 \2", text)
|
||||
text = re.sub(r"\(\s+", "(", text)
|
||||
text = re.sub(r"\s+\)", ")", text)
|
||||
return text.strip(" \t,;:")
|
||||
|
||||
|
||||
def normalize_block(block: str, block_number: int) -> tuple[str, Change | None]:
|
||||
original = block
|
||||
|
||||
result, removed_fillers = remove_fillers(original)
|
||||
result, duplicate_reductions = reduce_duplicate_phrases(result)
|
||||
result = tidy_spacing(result)
|
||||
|
||||
if result == original:
|
||||
return result, None
|
||||
|
||||
return result, Change(
|
||||
block_number=block_number,
|
||||
original=original,
|
||||
normalized=result,
|
||||
removed_fillers=removed_fillers,
|
||||
duplicate_reductions=duplicate_reductions,
|
||||
)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
|
||||
try:
|
||||
if not args.input_file.is_file():
|
||||
raise FileNotFoundError(f"Input file not found: {args.input_file}")
|
||||
|
||||
source = args.input_file.read_text(encoding="utf-8-sig")
|
||||
if not source.strip():
|
||||
raise ValueError("The input file is empty.")
|
||||
|
||||
output_path = args.output or args.input_file.with_name(
|
||||
f"{args.input_file.stem}_normalized.txt"
|
||||
)
|
||||
changes_path = args.changes or args.input_file.with_name(
|
||||
f"{args.input_file.stem}_changes.json"
|
||||
)
|
||||
|
||||
blocks = split_blocks(source)
|
||||
normalized_blocks: list[str] = []
|
||||
changes: list[Change] = []
|
||||
|
||||
for block_number, block in enumerate(blocks, start=1):
|
||||
normalized, change = normalize_block(block, block_number)
|
||||
normalized_blocks.append(normalized)
|
||||
if change is not None:
|
||||
changes.append(change)
|
||||
|
||||
normalized_text = "\n\n".join(normalized_blocks).strip() + "\n"
|
||||
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
changes_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
output_path.write_text(normalized_text, encoding="utf-8")
|
||||
|
||||
report = {
|
||||
"source_file": args.input_file.name,
|
||||
"output_file": output_path.name,
|
||||
"blocks_total": len(blocks),
|
||||
"blocks_changed": len(changes),
|
||||
"policy": {
|
||||
"removed": [
|
||||
"isolated filler sounds such as äh, ähm, hm, mhm",
|
||||
"immediate duplicate words",
|
||||
"immediate duplicate phrases of two to four words",
|
||||
"redundant whitespace",
|
||||
],
|
||||
"explicitly_preserved": [
|
||||
"negations",
|
||||
"modal words and qualifiers",
|
||||
"dates, times, numbers and quantities",
|
||||
"technical statements",
|
||||
"responsibilities",
|
||||
"deadlines",
|
||||
"decisions and commitments",
|
||||
],
|
||||
"principle": "When uncertain, leave the text unchanged.",
|
||||
},
|
||||
"changes": [asdict(change) for change in changes],
|
||||
}
|
||||
|
||||
changes_path.write_text(
|
||||
json.dumps(report, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
print(f"Source: {args.input_file}")
|
||||
print(f"Normalized: {output_path}")
|
||||
print(f"Change log: {changes_path}")
|
||||
print(f"Blocks total: {len(blocks)}")
|
||||
print(f"Blocks changed: {len(changes)}")
|
||||
return 0
|
||||
|
||||
except (OSError, UnicodeError, ValueError) as exc:
|
||||
print(f"Error: {exc}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
|
||||
@@ -0,0 +1,223 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Build a readable Markdown protocol from chunk extraction JSON files.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
EXTRACTION_CATEGORIES = (
|
||||
"facts",
|
||||
"decisions",
|
||||
"todos",
|
||||
"questions",
|
||||
"positions",
|
||||
"technical",
|
||||
)
|
||||
|
||||
CATEGORY_TITLES = {
|
||||
"facts": "Facts",
|
||||
"decisions": "Decisions",
|
||||
"todos": "Action Items",
|
||||
"questions": "Open Questions",
|
||||
"positions": "Positions",
|
||||
"technical": "Technical",
|
||||
}
|
||||
|
||||
CHUNK_NUMBER_RE = re.compile(r"chunk_(\d+)_")
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Build a Markdown meeting protocol from extraction JSON files."
|
||||
)
|
||||
parser.add_argument(
|
||||
"input_dir",
|
||||
type=Path,
|
||||
help="Directory containing chunk_XX_extraction.json files",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-o",
|
||||
"--output",
|
||||
type=Path,
|
||||
help="Output Markdown file; default: <input-dir>/meeting_protocol.md",
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def chunk_sort_key(path: Path) -> tuple[int, str]:
|
||||
match = CHUNK_NUMBER_RE.search(path.name)
|
||||
if not match:
|
||||
return (sys.maxsize, path.name)
|
||||
return (int(match.group(1)), path.name)
|
||||
|
||||
|
||||
def load_json(path: Path) -> Any:
|
||||
return json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
|
||||
|
||||
def find_extraction_files(input_dir: Path) -> list[Path]:
|
||||
return sorted(
|
||||
input_dir.glob("chunk_*_extraction.json"),
|
||||
key=chunk_sort_key,
|
||||
)
|
||||
|
||||
|
||||
def normalize_items(value: Any) -> list[str]:
|
||||
if not isinstance(value, list):
|
||||
return []
|
||||
|
||||
items: list[str] = []
|
||||
for item in value:
|
||||
if isinstance(item, str):
|
||||
text = item.strip()
|
||||
else:
|
||||
text = json.dumps(item, ensure_ascii=False, sort_keys=True)
|
||||
if text:
|
||||
items.append(text)
|
||||
return items
|
||||
|
||||
|
||||
def collect_extractions(
|
||||
extraction_files: list[Path],
|
||||
) -> dict[str, list[str]]:
|
||||
collected = {category: [] for category in EXTRACTION_CATEGORIES}
|
||||
|
||||
for path in extraction_files:
|
||||
data = load_json(path)
|
||||
if not isinstance(data, dict):
|
||||
raise ValueError(
|
||||
f"Extraction file must contain a JSON object: {path}"
|
||||
)
|
||||
|
||||
for category in EXTRACTION_CATEGORIES:
|
||||
collected[category].extend(
|
||||
normalize_items(data.get(category, []))
|
||||
)
|
||||
|
||||
return collected
|
||||
|
||||
|
||||
def find_topic_files(input_dir: Path) -> list[Path]:
|
||||
return sorted(
|
||||
input_dir.glob("chunk_*_normalized_windowed_segments.json"),
|
||||
key=chunk_sort_key,
|
||||
)
|
||||
|
||||
|
||||
def collect_topics(input_dir: Path) -> list[str]:
|
||||
topics: list[str] = []
|
||||
|
||||
for path in find_topic_files(input_dir):
|
||||
data = load_json(path)
|
||||
if not isinstance(data, dict):
|
||||
continue
|
||||
|
||||
source = data.get("source", {})
|
||||
source_file = (
|
||||
source.get("file")
|
||||
if isinstance(source, dict)
|
||||
else None
|
||||
)
|
||||
label = source_file or path.name
|
||||
|
||||
segments = data.get("segments", [])
|
||||
if not isinstance(segments, list):
|
||||
continue
|
||||
|
||||
for segment in segments:
|
||||
if not isinstance(segment, dict):
|
||||
continue
|
||||
segment_id = segment.get("segment_id", "segment")
|
||||
start_block = segment.get("start_block", "?")
|
||||
end_block = segment.get("end_block", "?")
|
||||
topics.append(
|
||||
f"{label}: {segment_id} (blocks {start_block}-{end_block})"
|
||||
)
|
||||
|
||||
return topics
|
||||
|
||||
|
||||
def render_list(items: list[str]) -> list[str]:
|
||||
if not items:
|
||||
return ["- None recorded."]
|
||||
return [f"- {item}" for item in items]
|
||||
|
||||
|
||||
def build_protocol_markdown(
|
||||
input_dir: Path,
|
||||
extraction_files: list[Path],
|
||||
) -> str:
|
||||
extractions = collect_extractions(extraction_files)
|
||||
topics = collect_topics(input_dir)
|
||||
|
||||
lines = [
|
||||
"# Meeting Protocol",
|
||||
"",
|
||||
"## Topics",
|
||||
*render_list(topics),
|
||||
"",
|
||||
"## Facts",
|
||||
*render_list(extractions["facts"]),
|
||||
"",
|
||||
"## Decisions",
|
||||
*render_list(extractions["decisions"]),
|
||||
"",
|
||||
"## Open Questions",
|
||||
*render_list(extractions["questions"]),
|
||||
"",
|
||||
"## Action Items",
|
||||
*render_list(extractions["todos"]),
|
||||
"",
|
||||
"## Positions",
|
||||
*render_list(extractions["positions"]),
|
||||
"",
|
||||
"## Technical",
|
||||
*render_list(extractions["technical"]),
|
||||
"",
|
||||
]
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def build_protocol(input_dir: Path, output_path: Path | None = None) -> Path:
|
||||
if not input_dir.is_dir():
|
||||
raise FileNotFoundError(f"Input directory not found: {input_dir}")
|
||||
|
||||
extraction_files = find_extraction_files(input_dir)
|
||||
if not extraction_files:
|
||||
raise ValueError(
|
||||
f"No extraction JSON files found in {input_dir}"
|
||||
)
|
||||
|
||||
output = output_path or input_dir / "meeting_protocol.md"
|
||||
markdown = build_protocol_markdown(input_dir, extraction_files)
|
||||
output.write_text(markdown, encoding="utf-8")
|
||||
return output
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
|
||||
try:
|
||||
extraction_files = find_extraction_files(args.input_dir)
|
||||
output_path = build_protocol(args.input_dir, args.output)
|
||||
except (OSError, UnicodeError, ValueError, json.JSONDecodeError) as exc:
|
||||
print(f"Error: {exc}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
print(f"Input: {args.input_dir}")
|
||||
print(f"Extraction JSON files: {len(extraction_files)}")
|
||||
print(f"Protocol: {output_path}")
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
|
||||
@@ -0,0 +1,124 @@
|
||||
import json
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
from src.meeting_lab.extraction.extract_chunks import (
|
||||
EXTRACTION_CATEGORIES,
|
||||
extraction_path_for_chunk,
|
||||
normalize_current_schema,
|
||||
parse_json_response,
|
||||
)
|
||||
from src.meeting_lab.protocol.build_protocol import build_protocol
|
||||
|
||||
|
||||
class ExtractionProtocolTests(unittest.TestCase):
|
||||
def test_extraction_path_drops_normalized_suffix(self) -> None:
|
||||
path = Path("chunks/chunk_01_normalized.txt")
|
||||
|
||||
self.assertEqual(
|
||||
extraction_path_for_chunk(path),
|
||||
Path("chunks/chunk_01_extraction.json"),
|
||||
)
|
||||
|
||||
def test_normalize_current_schema_maps_legacy_response(self) -> None:
|
||||
result = normalize_current_schema(
|
||||
{
|
||||
"facts": [
|
||||
{
|
||||
"speaker": "A",
|
||||
"statement": "Fact one",
|
||||
"status": "clear",
|
||||
"evidence": "Fact",
|
||||
}
|
||||
],
|
||||
"decisions": [
|
||||
{
|
||||
"decision": "Decision one",
|
||||
"evidence": "Decision",
|
||||
}
|
||||
],
|
||||
"todos": [
|
||||
{
|
||||
"task": "Todo one",
|
||||
"responsible": "B",
|
||||
"deadline": None,
|
||||
"evidence": "Todo",
|
||||
}
|
||||
],
|
||||
"open_questions": [
|
||||
{
|
||||
"question": "Question one",
|
||||
"evidence": "Question",
|
||||
}
|
||||
],
|
||||
"technical_details": [
|
||||
{
|
||||
"subject": "System",
|
||||
"statement": "Technical one",
|
||||
"status": "clear",
|
||||
"evidence": "Technical",
|
||||
}
|
||||
],
|
||||
}
|
||||
)
|
||||
|
||||
self.assertEqual(set(result), set(EXTRACTION_CATEGORIES))
|
||||
self.assertEqual(result["positions"], [])
|
||||
self.assertIn("Fact one", result["facts"][0])
|
||||
self.assertIn("Decision one", result["decisions"][0])
|
||||
self.assertIn("Todo one", result["todos"][0])
|
||||
self.assertIn("Question one", result["questions"][0])
|
||||
self.assertIn("Technical one", result["technical"][0])
|
||||
|
||||
def test_parse_json_response_uses_final_object_after_thinking(self) -> None:
|
||||
text = """
|
||||
Thinking:
|
||||
I will reason about the transcript first.
|
||||
{"facts": ["draft"], "decisions": []}
|
||||
|
||||
Final answer:
|
||||
{
|
||||
"facts": ["final"],
|
||||
"decisions": [],
|
||||
"todos": [],
|
||||
"questions": [],
|
||||
"positions": [],
|
||||
"technical": []
|
||||
}
|
||||
"""
|
||||
|
||||
parsed = parse_json_response(text)
|
||||
|
||||
self.assertEqual(parsed["facts"], ["final"])
|
||||
self.assertEqual(set(parsed), set(EXTRACTION_CATEGORIES))
|
||||
|
||||
def test_build_protocol_groups_extraction_items(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
input_dir = Path(directory)
|
||||
(input_dir / "chunk_01_extraction.json").write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"facts": ["Fact one"],
|
||||
"decisions": ["Decision one"],
|
||||
"todos": ["Todo one"],
|
||||
"questions": ["Question one"],
|
||||
"positions": [],
|
||||
"technical": [],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
output_path = build_protocol(input_dir)
|
||||
|
||||
markdown = output_path.read_text(encoding="utf-8")
|
||||
self.assertIn("# Meeting Protocol", markdown)
|
||||
self.assertIn("## Facts\n- Fact one", markdown)
|
||||
self.assertIn("## Decisions\n- Decision one", markdown)
|
||||
self.assertIn("## Open Questions\n- Question one", markdown)
|
||||
self.assertIn("## Action Items\n- Todo one", markdown)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user