Add windowed segmentation pipeline and review tooling

This commit is contained in:
2026-07-22 16:15:02 +02:00
parent 889a4fe32d
commit 1a6d731d21
10 changed files with 9364 additions and 1 deletions
@@ -0,0 +1,653 @@
#!/usr/bin/env python3
"""
Create a human-readable review report for transcript segmentation results.
The tool reads:
1. a normalized transcript text file
2. a segmentation JSON file
It recreates the analysis blocks using the same block-size setting stored in
the segmentation result and writes a Markdown report containing:
- all generated segments
- the complete text of every segment
- a compact review section around every detected boundary
Example:
python src/meeting_lab/segmentation/review_segmentation.py \
samples/chunks/chunk_01_normalized.txt \
samples/chunks/chunk_01_normalized_windowed_segments.json
"""
from __future__ import annotations
import argparse
import json
import re
import sys
from pathlib import Path
from typing import Any
DEFAULT_CONTEXT_BLOCKS = 2
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description=(
"Create a readable review report for transcript segmentation."
)
)
parser.add_argument(
"transcript_file",
type=Path,
help="Normalized transcript text file",
)
parser.add_argument(
"segmentation_file",
type=Path,
help="Segmentation result JSON file",
)
parser.add_argument(
"-o",
"--output",
type=Path,
help=(
"Output Markdown file; default: "
"<segmentation_file>_review.md"
),
)
parser.add_argument(
"--context-blocks",
type=int,
default=DEFAULT_CONTEXT_BLOCKS,
help=(
"Number of blocks shown before and after each boundary "
f"(default: {DEFAULT_CONTEXT_BLOCKS})"
),
)
return parser.parse_args()
def split_into_blocks(
text: str,
target_chars: int,
) -> list[str]:
"""
Recreate the same analysis blocks used by the segmentation tool.
"""
if target_chars < 1:
raise ValueError(
"target_chars must be greater than zero."
)
normalized = (
text.replace("\r\n", "\n")
.replace("\r", "\n")
.strip()
)
units = [
part.strip()
for part in re.split(r"\n\s*\n+", normalized)
if part.strip()
]
if not units:
units = [
line.strip()
for line in normalized.splitlines()
if line.strip()
]
blocks: list[str] = []
current_units: list[str] = []
current_length = 0
for unit in units:
separator_length = 1 if current_units else 0
projected_length = (
current_length
+ separator_length
+ len(unit)
)
if current_units and projected_length > target_chars:
blocks.append(
"\n".join(current_units)
)
current_units = [unit]
current_length = len(unit)
else:
current_units.append(unit)
current_length = projected_length
if current_units:
blocks.append(
"\n".join(current_units)
)
return blocks
def load_json_file(
path: Path,
) -> dict[str, Any]:
try:
content = path.read_text(
encoding="utf-8-sig"
)
data = json.loads(content)
except json.JSONDecodeError as exc:
raise ValueError(
f"Invalid JSON in {path}: {exc}"
) from exc
if not isinstance(data, dict):
raise ValueError(
"The segmentation file must contain a JSON object."
)
return data
def get_target_block_chars(
segmentation: dict[str, Any],
) -> int:
source = segmentation.get("source")
if not isinstance(source, dict):
raise ValueError(
"The segmentation JSON contains no valid source object."
)
target_chars = source.get("target_block_chars")
if not isinstance(target_chars, int):
raise ValueError(
"The segmentation JSON contains no valid "
"source.target_block_chars value."
)
if target_chars < 1:
raise ValueError(
"source.target_block_chars must be greater than zero."
)
return target_chars
def get_expected_block_count(
segmentation: dict[str, Any],
) -> int:
source = segmentation.get("source")
if not isinstance(source, dict):
raise ValueError(
"The segmentation JSON contains no valid source object."
)
block_count = source.get("block_count")
if not isinstance(block_count, int):
raise ValueError(
"The segmentation JSON contains no valid "
"source.block_count value."
)
if block_count < 1:
raise ValueError(
"source.block_count must be greater than zero."
)
return block_count
def get_segments(
segmentation: dict[str, Any],
) -> list[dict[str, Any]]:
raw_segments = segmentation.get("segments")
if not isinstance(raw_segments, list):
raise ValueError(
"The segmentation JSON contains no segments list."
)
segments: list[dict[str, Any]] = []
for index, segment in enumerate(
raw_segments,
start=1,
):
if not isinstance(segment, dict):
raise ValueError(
f"Segment {index} is not a JSON object."
)
segment_id = segment.get("segment_id")
start_block = segment.get("start_block")
end_block = segment.get("end_block")
if not isinstance(segment_id, str):
raise ValueError(
f"Segment {index} has no valid segment_id."
)
if not isinstance(start_block, int):
raise ValueError(
f"Segment {segment_id} has no valid start_block."
)
if not isinstance(end_block, int):
raise ValueError(
f"Segment {segment_id} has no valid end_block."
)
if start_block < 1:
raise ValueError(
f"Segment {segment_id} starts before block 1."
)
if end_block < start_block:
raise ValueError(
f"Segment {segment_id} ends before it starts."
)
segments.append(
{
"segment_id": segment_id,
"start_block": start_block,
"end_block": end_block,
}
)
return segments
def validate_segments(
segments: list[dict[str, Any]],
block_count: int,
) -> None:
if not segments:
raise ValueError(
"The segmentation result contains no segments."
)
expected_start = 1
for segment in segments:
start_block = segment["start_block"]
end_block = segment["end_block"]
segment_id = segment["segment_id"]
if start_block != expected_start:
raise ValueError(
f"Segment {segment_id} starts at block "
f"{start_block}; expected block {expected_start}."
)
if end_block > block_count:
raise ValueError(
f"Segment {segment_id} ends after the final block."
)
expected_start = end_block + 1
if expected_start != block_count + 1:
raise ValueError(
"The segments do not cover all transcript blocks."
)
def render_block(
block_number: int,
block_text: str,
) -> str:
return (
f"### Block {block_number}\n\n"
f"{block_text}\n"
)
def render_segment(
segment: dict[str, Any],
blocks: list[str],
) -> str:
segment_id = segment["segment_id"]
start_block = segment["start_block"]
end_block = segment["end_block"]
block_span = end_block - start_block + 1
lines = [
f"## {segment_id}",
"",
f"**Blöcke:** {start_block}–{end_block}",
"",
f"**Umfang:** {block_span} Blöcke",
"",
]
for block_number in range(
start_block,
end_block + 1,
):
lines.append(
render_block(
block_number=block_number,
block_text=blocks[block_number - 1],
)
)
return "\n".join(lines)
def render_boundary_review(
boundary: int,
blocks: list[str],
context_blocks: int,
) -> str:
block_count = len(blocks)
before_start = max(
1,
boundary - context_blocks + 1,
)
after_end = min(
block_count,
boundary + context_blocks,
)
lines = [
f"## Themenwechsel nach Block {boundary}",
"",
f"Der nächste Abschnitt beginnt mit Block {boundary + 1}.",
"",
"### Vor der Grenze",
"",
]
for block_number in range(
before_start,
boundary + 1,
):
lines.append(
render_block(
block_number=block_number,
block_text=blocks[block_number - 1],
)
)
lines.extend(
[
"",
"---",
"",
"## ⟶ Erkannter Themenwechsel",
"",
"---",
"",
"### Nach der Grenze",
"",
]
)
for block_number in range(
boundary + 1,
after_end + 1,
):
lines.append(
render_block(
block_number=block_number,
block_text=blocks[block_number - 1],
)
)
lines.extend(
[
"",
"### Manuelle Bewertung",
"",
"- [ ] echter Themenwechsel",
"- [ ] nur Unterthema",
"- [ ] kurze Abschweifung",
"- [ ] kein Themenwechsel",
"",
"**Notiz:**",
"",
"",
]
)
return "\n".join(lines)
def build_report(
transcript_file: Path,
segmentation_file: Path,
segmentation: dict[str, Any],
blocks: list[str],
segments: list[dict[str, Any]],
context_blocks: int,
) -> str:
configuration = segmentation.get(
"configuration",
{},
)
if not isinstance(configuration, dict):
configuration = {}
model = configuration.get(
"model",
"unbekannt",
)
window_size = configuration.get(
"window_size",
"unbekannt",
)
window_overlap = configuration.get(
"window_overlap",
"unbekannt",
)
boundaries = [
segment["end_block"]
for segment in segments[:-1]
]
lines = [
"# Review der Meeting-Segmentierung",
"",
"## Metadaten",
"",
f"- **Transkript:** `{transcript_file}`",
f"- **Segmentierung:** `{segmentation_file}`",
f"- **Modell:** `{model}`",
f"- **Blöcke:** {len(blocks)}",
f"- **Segmente:** {len(segments)}",
f"- **Themengrenzen:** {len(boundaries)}",
f"- **Fenstergröße:** {window_size}",
f"- **Überlappung:** {window_overlap}",
"",
"## Erkannte Grenzen",
"",
(
", ".join(str(value) for value in boundaries)
if boundaries
else "Keine"
),
"",
"---",
"",
"# Grenzprüfung",
"",
]
for boundary in boundaries:
lines.append(
render_boundary_review(
boundary=boundary,
blocks=blocks,
context_blocks=context_blocks,
)
)
lines.extend(
[
"",
"---",
"",
]
)
lines.extend(
[
"# Vollständige Segmente",
"",
]
)
for segment in segments:
lines.append(
render_segment(
segment=segment,
blocks=blocks,
)
)
lines.extend(
[
"",
"---",
"",
]
)
return "\n".join(lines).rstrip() + "\n"
def main() -> int:
args = parse_args()
try:
if not args.transcript_file.is_file():
raise FileNotFoundError(
f"Transcript file not found: "
f"{args.transcript_file}"
)
if not args.segmentation_file.is_file():
raise FileNotFoundError(
f"Segmentation file not found: "
f"{args.segmentation_file}"
)
if args.context_blocks < 1:
raise ValueError(
"context_blocks must be at least 1."
)
transcript = args.transcript_file.read_text(
encoding="utf-8-sig"
).strip()
if not transcript:
raise ValueError(
"The transcript file is empty."
)
segmentation = load_json_file(
args.segmentation_file
)
target_block_chars = get_target_block_chars(
segmentation
)
expected_block_count = get_expected_block_count(
segmentation
)
blocks = split_into_blocks(
text=transcript,
target_chars=target_block_chars,
)
if len(blocks) != expected_block_count:
raise ValueError(
"The recreated block count does not match the "
"segmentation result: "
f"{len(blocks)} instead of "
f"{expected_block_count}."
)
segments = get_segments(
segmentation
)
validate_segments(
segments=segments,
block_count=len(blocks),
)
output_path = (
args.output
or args.segmentation_file.with_name(
f"{args.segmentation_file.stem}_review.md"
)
)
output_path.parent.mkdir(
parents=True,
exist_ok=True,
)
report = build_report(
transcript_file=args.transcript_file,
segmentation_file=args.segmentation_file,
segmentation=segmentation,
blocks=blocks,
segments=segments,
context_blocks=args.context_blocks,
)
output_path.write_text(
report,
encoding="utf-8",
)
print(f"Transcript: {args.transcript_file}")
print(f"Segments: {args.segmentation_file}")
print(f"Blocks: {len(blocks)}")
print(f"Boundaries: {len(segments) - 1}")
print(f"Output: {output_path}")
return 0
except (
OSError,
UnicodeError,
ValueError,
) as exc:
print(
f"Error: {exc}",
file=sys.stderr,
)
return 1
if __name__ == "__main__":
raise SystemExit(main())