Add windowed segmentation pipeline and review tooling
This commit is contained in:
@@ -0,0 +1,653 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Create a human-readable review report for transcript segmentation results.
|
||||
|
||||
The tool reads:
|
||||
|
||||
1. a normalized transcript text file
|
||||
2. a segmentation JSON file
|
||||
|
||||
It recreates the analysis blocks using the same block-size setting stored in
|
||||
the segmentation result and writes a Markdown report containing:
|
||||
|
||||
- all generated segments
|
||||
- the complete text of every segment
|
||||
- a compact review section around every detected boundary
|
||||
|
||||
Example:
|
||||
|
||||
python src/meeting_lab/segmentation/review_segmentation.py \
|
||||
samples/chunks/chunk_01_normalized.txt \
|
||||
samples/chunks/chunk_01_normalized_windowed_segments.json
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
DEFAULT_CONTEXT_BLOCKS = 2
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description=(
|
||||
"Create a readable review report for transcript segmentation."
|
||||
)
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
"transcript_file",
|
||||
type=Path,
|
||||
help="Normalized transcript text file",
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
"segmentation_file",
|
||||
type=Path,
|
||||
help="Segmentation result JSON file",
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
"-o",
|
||||
"--output",
|
||||
type=Path,
|
||||
help=(
|
||||
"Output Markdown file; default: "
|
||||
"<segmentation_file>_review.md"
|
||||
),
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
"--context-blocks",
|
||||
type=int,
|
||||
default=DEFAULT_CONTEXT_BLOCKS,
|
||||
help=(
|
||||
"Number of blocks shown before and after each boundary "
|
||||
f"(default: {DEFAULT_CONTEXT_BLOCKS})"
|
||||
),
|
||||
)
|
||||
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def split_into_blocks(
|
||||
text: str,
|
||||
target_chars: int,
|
||||
) -> list[str]:
|
||||
"""
|
||||
Recreate the same analysis blocks used by the segmentation tool.
|
||||
"""
|
||||
if target_chars < 1:
|
||||
raise ValueError(
|
||||
"target_chars must be greater than zero."
|
||||
)
|
||||
|
||||
normalized = (
|
||||
text.replace("\r\n", "\n")
|
||||
.replace("\r", "\n")
|
||||
.strip()
|
||||
)
|
||||
|
||||
units = [
|
||||
part.strip()
|
||||
for part in re.split(r"\n\s*\n+", normalized)
|
||||
if part.strip()
|
||||
]
|
||||
|
||||
if not units:
|
||||
units = [
|
||||
line.strip()
|
||||
for line in normalized.splitlines()
|
||||
if line.strip()
|
||||
]
|
||||
|
||||
blocks: list[str] = []
|
||||
current_units: list[str] = []
|
||||
current_length = 0
|
||||
|
||||
for unit in units:
|
||||
separator_length = 1 if current_units else 0
|
||||
|
||||
projected_length = (
|
||||
current_length
|
||||
+ separator_length
|
||||
+ len(unit)
|
||||
)
|
||||
|
||||
if current_units and projected_length > target_chars:
|
||||
blocks.append(
|
||||
"\n".join(current_units)
|
||||
)
|
||||
|
||||
current_units = [unit]
|
||||
current_length = len(unit)
|
||||
|
||||
else:
|
||||
current_units.append(unit)
|
||||
current_length = projected_length
|
||||
|
||||
if current_units:
|
||||
blocks.append(
|
||||
"\n".join(current_units)
|
||||
)
|
||||
|
||||
return blocks
|
||||
|
||||
|
||||
def load_json_file(
|
||||
path: Path,
|
||||
) -> dict[str, Any]:
|
||||
try:
|
||||
content = path.read_text(
|
||||
encoding="utf-8-sig"
|
||||
)
|
||||
data = json.loads(content)
|
||||
|
||||
except json.JSONDecodeError as exc:
|
||||
raise ValueError(
|
||||
f"Invalid JSON in {path}: {exc}"
|
||||
) from exc
|
||||
|
||||
if not isinstance(data, dict):
|
||||
raise ValueError(
|
||||
"The segmentation file must contain a JSON object."
|
||||
)
|
||||
|
||||
return data
|
||||
|
||||
|
||||
def get_target_block_chars(
|
||||
segmentation: dict[str, Any],
|
||||
) -> int:
|
||||
source = segmentation.get("source")
|
||||
|
||||
if not isinstance(source, dict):
|
||||
raise ValueError(
|
||||
"The segmentation JSON contains no valid source object."
|
||||
)
|
||||
|
||||
target_chars = source.get("target_block_chars")
|
||||
|
||||
if not isinstance(target_chars, int):
|
||||
raise ValueError(
|
||||
"The segmentation JSON contains no valid "
|
||||
"source.target_block_chars value."
|
||||
)
|
||||
|
||||
if target_chars < 1:
|
||||
raise ValueError(
|
||||
"source.target_block_chars must be greater than zero."
|
||||
)
|
||||
|
||||
return target_chars
|
||||
|
||||
|
||||
def get_expected_block_count(
|
||||
segmentation: dict[str, Any],
|
||||
) -> int:
|
||||
source = segmentation.get("source")
|
||||
|
||||
if not isinstance(source, dict):
|
||||
raise ValueError(
|
||||
"The segmentation JSON contains no valid source object."
|
||||
)
|
||||
|
||||
block_count = source.get("block_count")
|
||||
|
||||
if not isinstance(block_count, int):
|
||||
raise ValueError(
|
||||
"The segmentation JSON contains no valid "
|
||||
"source.block_count value."
|
||||
)
|
||||
|
||||
if block_count < 1:
|
||||
raise ValueError(
|
||||
"source.block_count must be greater than zero."
|
||||
)
|
||||
|
||||
return block_count
|
||||
|
||||
|
||||
def get_segments(
|
||||
segmentation: dict[str, Any],
|
||||
) -> list[dict[str, Any]]:
|
||||
raw_segments = segmentation.get("segments")
|
||||
|
||||
if not isinstance(raw_segments, list):
|
||||
raise ValueError(
|
||||
"The segmentation JSON contains no segments list."
|
||||
)
|
||||
|
||||
segments: list[dict[str, Any]] = []
|
||||
|
||||
for index, segment in enumerate(
|
||||
raw_segments,
|
||||
start=1,
|
||||
):
|
||||
if not isinstance(segment, dict):
|
||||
raise ValueError(
|
||||
f"Segment {index} is not a JSON object."
|
||||
)
|
||||
|
||||
segment_id = segment.get("segment_id")
|
||||
start_block = segment.get("start_block")
|
||||
end_block = segment.get("end_block")
|
||||
|
||||
if not isinstance(segment_id, str):
|
||||
raise ValueError(
|
||||
f"Segment {index} has no valid segment_id."
|
||||
)
|
||||
|
||||
if not isinstance(start_block, int):
|
||||
raise ValueError(
|
||||
f"Segment {segment_id} has no valid start_block."
|
||||
)
|
||||
|
||||
if not isinstance(end_block, int):
|
||||
raise ValueError(
|
||||
f"Segment {segment_id} has no valid end_block."
|
||||
)
|
||||
|
||||
if start_block < 1:
|
||||
raise ValueError(
|
||||
f"Segment {segment_id} starts before block 1."
|
||||
)
|
||||
|
||||
if end_block < start_block:
|
||||
raise ValueError(
|
||||
f"Segment {segment_id} ends before it starts."
|
||||
)
|
||||
|
||||
segments.append(
|
||||
{
|
||||
"segment_id": segment_id,
|
||||
"start_block": start_block,
|
||||
"end_block": end_block,
|
||||
}
|
||||
)
|
||||
|
||||
return segments
|
||||
|
||||
|
||||
def validate_segments(
|
||||
segments: list[dict[str, Any]],
|
||||
block_count: int,
|
||||
) -> None:
|
||||
if not segments:
|
||||
raise ValueError(
|
||||
"The segmentation result contains no segments."
|
||||
)
|
||||
|
||||
expected_start = 1
|
||||
|
||||
for segment in segments:
|
||||
start_block = segment["start_block"]
|
||||
end_block = segment["end_block"]
|
||||
segment_id = segment["segment_id"]
|
||||
|
||||
if start_block != expected_start:
|
||||
raise ValueError(
|
||||
f"Segment {segment_id} starts at block "
|
||||
f"{start_block}; expected block {expected_start}."
|
||||
)
|
||||
|
||||
if end_block > block_count:
|
||||
raise ValueError(
|
||||
f"Segment {segment_id} ends after the final block."
|
||||
)
|
||||
|
||||
expected_start = end_block + 1
|
||||
|
||||
if expected_start != block_count + 1:
|
||||
raise ValueError(
|
||||
"The segments do not cover all transcript blocks."
|
||||
)
|
||||
|
||||
|
||||
def render_block(
|
||||
block_number: int,
|
||||
block_text: str,
|
||||
) -> str:
|
||||
return (
|
||||
f"### Block {block_number}\n\n"
|
||||
f"{block_text}\n"
|
||||
)
|
||||
|
||||
|
||||
def render_segment(
|
||||
segment: dict[str, Any],
|
||||
blocks: list[str],
|
||||
) -> str:
|
||||
segment_id = segment["segment_id"]
|
||||
start_block = segment["start_block"]
|
||||
end_block = segment["end_block"]
|
||||
block_span = end_block - start_block + 1
|
||||
|
||||
lines = [
|
||||
f"## {segment_id}",
|
||||
"",
|
||||
f"**Blöcke:** {start_block}–{end_block}",
|
||||
"",
|
||||
f"**Umfang:** {block_span} Blöcke",
|
||||
"",
|
||||
]
|
||||
|
||||
for block_number in range(
|
||||
start_block,
|
||||
end_block + 1,
|
||||
):
|
||||
lines.append(
|
||||
render_block(
|
||||
block_number=block_number,
|
||||
block_text=blocks[block_number - 1],
|
||||
)
|
||||
)
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def render_boundary_review(
|
||||
boundary: int,
|
||||
blocks: list[str],
|
||||
context_blocks: int,
|
||||
) -> str:
|
||||
block_count = len(blocks)
|
||||
|
||||
before_start = max(
|
||||
1,
|
||||
boundary - context_blocks + 1,
|
||||
)
|
||||
after_end = min(
|
||||
block_count,
|
||||
boundary + context_blocks,
|
||||
)
|
||||
|
||||
lines = [
|
||||
f"## Themenwechsel nach Block {boundary}",
|
||||
"",
|
||||
f"Der nächste Abschnitt beginnt mit Block {boundary + 1}.",
|
||||
"",
|
||||
"### Vor der Grenze",
|
||||
"",
|
||||
]
|
||||
|
||||
for block_number in range(
|
||||
before_start,
|
||||
boundary + 1,
|
||||
):
|
||||
lines.append(
|
||||
render_block(
|
||||
block_number=block_number,
|
||||
block_text=blocks[block_number - 1],
|
||||
)
|
||||
)
|
||||
|
||||
lines.extend(
|
||||
[
|
||||
"",
|
||||
"---",
|
||||
"",
|
||||
"## ⟶ Erkannter Themenwechsel",
|
||||
"",
|
||||
"---",
|
||||
"",
|
||||
"### Nach der Grenze",
|
||||
"",
|
||||
]
|
||||
)
|
||||
|
||||
for block_number in range(
|
||||
boundary + 1,
|
||||
after_end + 1,
|
||||
):
|
||||
lines.append(
|
||||
render_block(
|
||||
block_number=block_number,
|
||||
block_text=blocks[block_number - 1],
|
||||
)
|
||||
)
|
||||
|
||||
lines.extend(
|
||||
[
|
||||
"",
|
||||
"### Manuelle Bewertung",
|
||||
"",
|
||||
"- [ ] echter Themenwechsel",
|
||||
"- [ ] nur Unterthema",
|
||||
"- [ ] kurze Abschweifung",
|
||||
"- [ ] kein Themenwechsel",
|
||||
"",
|
||||
"**Notiz:**",
|
||||
"",
|
||||
"",
|
||||
]
|
||||
)
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def build_report(
|
||||
transcript_file: Path,
|
||||
segmentation_file: Path,
|
||||
segmentation: dict[str, Any],
|
||||
blocks: list[str],
|
||||
segments: list[dict[str, Any]],
|
||||
context_blocks: int,
|
||||
) -> str:
|
||||
configuration = segmentation.get(
|
||||
"configuration",
|
||||
{},
|
||||
)
|
||||
|
||||
if not isinstance(configuration, dict):
|
||||
configuration = {}
|
||||
|
||||
model = configuration.get(
|
||||
"model",
|
||||
"unbekannt",
|
||||
)
|
||||
|
||||
window_size = configuration.get(
|
||||
"window_size",
|
||||
"unbekannt",
|
||||
)
|
||||
|
||||
window_overlap = configuration.get(
|
||||
"window_overlap",
|
||||
"unbekannt",
|
||||
)
|
||||
|
||||
boundaries = [
|
||||
segment["end_block"]
|
||||
for segment in segments[:-1]
|
||||
]
|
||||
|
||||
lines = [
|
||||
"# Review der Meeting-Segmentierung",
|
||||
"",
|
||||
"## Metadaten",
|
||||
"",
|
||||
f"- **Transkript:** `{transcript_file}`",
|
||||
f"- **Segmentierung:** `{segmentation_file}`",
|
||||
f"- **Modell:** `{model}`",
|
||||
f"- **Blöcke:** {len(blocks)}",
|
||||
f"- **Segmente:** {len(segments)}",
|
||||
f"- **Themengrenzen:** {len(boundaries)}",
|
||||
f"- **Fenstergröße:** {window_size}",
|
||||
f"- **Überlappung:** {window_overlap}",
|
||||
"",
|
||||
"## Erkannte Grenzen",
|
||||
"",
|
||||
(
|
||||
", ".join(str(value) for value in boundaries)
|
||||
if boundaries
|
||||
else "Keine"
|
||||
),
|
||||
"",
|
||||
"---",
|
||||
"",
|
||||
"# Grenzprüfung",
|
||||
"",
|
||||
]
|
||||
|
||||
for boundary in boundaries:
|
||||
lines.append(
|
||||
render_boundary_review(
|
||||
boundary=boundary,
|
||||
blocks=blocks,
|
||||
context_blocks=context_blocks,
|
||||
)
|
||||
)
|
||||
|
||||
lines.extend(
|
||||
[
|
||||
"",
|
||||
"---",
|
||||
"",
|
||||
]
|
||||
)
|
||||
|
||||
lines.extend(
|
||||
[
|
||||
"# Vollständige Segmente",
|
||||
"",
|
||||
]
|
||||
)
|
||||
|
||||
for segment in segments:
|
||||
lines.append(
|
||||
render_segment(
|
||||
segment=segment,
|
||||
blocks=blocks,
|
||||
)
|
||||
)
|
||||
|
||||
lines.extend(
|
||||
[
|
||||
"",
|
||||
"---",
|
||||
"",
|
||||
]
|
||||
)
|
||||
|
||||
return "\n".join(lines).rstrip() + "\n"
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
|
||||
try:
|
||||
if not args.transcript_file.is_file():
|
||||
raise FileNotFoundError(
|
||||
f"Transcript file not found: "
|
||||
f"{args.transcript_file}"
|
||||
)
|
||||
|
||||
if not args.segmentation_file.is_file():
|
||||
raise FileNotFoundError(
|
||||
f"Segmentation file not found: "
|
||||
f"{args.segmentation_file}"
|
||||
)
|
||||
|
||||
if args.context_blocks < 1:
|
||||
raise ValueError(
|
||||
"context_blocks must be at least 1."
|
||||
)
|
||||
|
||||
transcript = args.transcript_file.read_text(
|
||||
encoding="utf-8-sig"
|
||||
).strip()
|
||||
|
||||
if not transcript:
|
||||
raise ValueError(
|
||||
"The transcript file is empty."
|
||||
)
|
||||
|
||||
segmentation = load_json_file(
|
||||
args.segmentation_file
|
||||
)
|
||||
|
||||
target_block_chars = get_target_block_chars(
|
||||
segmentation
|
||||
)
|
||||
|
||||
expected_block_count = get_expected_block_count(
|
||||
segmentation
|
||||
)
|
||||
|
||||
blocks = split_into_blocks(
|
||||
text=transcript,
|
||||
target_chars=target_block_chars,
|
||||
)
|
||||
|
||||
if len(blocks) != expected_block_count:
|
||||
raise ValueError(
|
||||
"The recreated block count does not match the "
|
||||
"segmentation result: "
|
||||
f"{len(blocks)} instead of "
|
||||
f"{expected_block_count}."
|
||||
)
|
||||
|
||||
segments = get_segments(
|
||||
segmentation
|
||||
)
|
||||
|
||||
validate_segments(
|
||||
segments=segments,
|
||||
block_count=len(blocks),
|
||||
)
|
||||
|
||||
output_path = (
|
||||
args.output
|
||||
or args.segmentation_file.with_name(
|
||||
f"{args.segmentation_file.stem}_review.md"
|
||||
)
|
||||
)
|
||||
|
||||
output_path.parent.mkdir(
|
||||
parents=True,
|
||||
exist_ok=True,
|
||||
)
|
||||
|
||||
report = build_report(
|
||||
transcript_file=args.transcript_file,
|
||||
segmentation_file=args.segmentation_file,
|
||||
segmentation=segmentation,
|
||||
blocks=blocks,
|
||||
segments=segments,
|
||||
context_blocks=args.context_blocks,
|
||||
)
|
||||
|
||||
output_path.write_text(
|
||||
report,
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
print(f"Transcript: {args.transcript_file}")
|
||||
print(f"Segments: {args.segmentation_file}")
|
||||
print(f"Blocks: {len(blocks)}")
|
||||
print(f"Boundaries: {len(segments) - 1}")
|
||||
print(f"Output: {output_path}")
|
||||
|
||||
return 0
|
||||
|
||||
except (
|
||||
OSError,
|
||||
UnicodeError,
|
||||
ValueError,
|
||||
) as exc:
|
||||
print(
|
||||
f"Error: {exc}",
|
||||
file=sys.stderr,
|
||||
)
|
||||
return 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user