654 lines
15 KiB
Python
654 lines
15 KiB
Python
#!/usr/bin/env python3
|
||
"""
|
||
Create a human-readable review report for transcript segmentation results.
|
||
|
||
The tool reads:
|
||
|
||
1. a normalized transcript text file
|
||
2. a segmentation JSON file
|
||
|
||
It recreates the analysis blocks using the same block-size setting stored in
|
||
the segmentation result and writes a Markdown report containing:
|
||
|
||
- all generated segments
|
||
- the complete text of every segment
|
||
- a compact review section around every detected boundary
|
||
|
||
Example:
|
||
|
||
python src/meeting_lab/segmentation/review_segmentation.py \
|
||
samples/chunks/chunk_01_normalized.txt \
|
||
samples/chunks/chunk_01_normalized_windowed_segments.json
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import json
|
||
import re
|
||
import sys
|
||
from pathlib import Path
|
||
from typing import Any
|
||
|
||
|
||
DEFAULT_CONTEXT_BLOCKS = 2
|
||
|
||
|
||
def parse_args() -> argparse.Namespace:
|
||
parser = argparse.ArgumentParser(
|
||
description=(
|
||
"Create a readable review report for transcript segmentation."
|
||
)
|
||
)
|
||
|
||
parser.add_argument(
|
||
"transcript_file",
|
||
type=Path,
|
||
help="Normalized transcript text file",
|
||
)
|
||
|
||
parser.add_argument(
|
||
"segmentation_file",
|
||
type=Path,
|
||
help="Segmentation result JSON file",
|
||
)
|
||
|
||
parser.add_argument(
|
||
"-o",
|
||
"--output",
|
||
type=Path,
|
||
help=(
|
||
"Output Markdown file; default: "
|
||
"<segmentation_file>_review.md"
|
||
),
|
||
)
|
||
|
||
parser.add_argument(
|
||
"--context-blocks",
|
||
type=int,
|
||
default=DEFAULT_CONTEXT_BLOCKS,
|
||
help=(
|
||
"Number of blocks shown before and after each boundary "
|
||
f"(default: {DEFAULT_CONTEXT_BLOCKS})"
|
||
),
|
||
)
|
||
|
||
return parser.parse_args()
|
||
|
||
|
||
def split_into_blocks(
|
||
text: str,
|
||
target_chars: int,
|
||
) -> list[str]:
|
||
"""
|
||
Recreate the same analysis blocks used by the segmentation tool.
|
||
"""
|
||
if target_chars < 1:
|
||
raise ValueError(
|
||
"target_chars must be greater than zero."
|
||
)
|
||
|
||
normalized = (
|
||
text.replace("\r\n", "\n")
|
||
.replace("\r", "\n")
|
||
.strip()
|
||
)
|
||
|
||
units = [
|
||
part.strip()
|
||
for part in re.split(r"\n\s*\n+", normalized)
|
||
if part.strip()
|
||
]
|
||
|
||
if not units:
|
||
units = [
|
||
line.strip()
|
||
for line in normalized.splitlines()
|
||
if line.strip()
|
||
]
|
||
|
||
blocks: list[str] = []
|
||
current_units: list[str] = []
|
||
current_length = 0
|
||
|
||
for unit in units:
|
||
separator_length = 1 if current_units else 0
|
||
|
||
projected_length = (
|
||
current_length
|
||
+ separator_length
|
||
+ len(unit)
|
||
)
|
||
|
||
if current_units and projected_length > target_chars:
|
||
blocks.append(
|
||
"\n".join(current_units)
|
||
)
|
||
|
||
current_units = [unit]
|
||
current_length = len(unit)
|
||
|
||
else:
|
||
current_units.append(unit)
|
||
current_length = projected_length
|
||
|
||
if current_units:
|
||
blocks.append(
|
||
"\n".join(current_units)
|
||
)
|
||
|
||
return blocks
|
||
|
||
|
||
def load_json_file(
|
||
path: Path,
|
||
) -> dict[str, Any]:
|
||
try:
|
||
content = path.read_text(
|
||
encoding="utf-8-sig"
|
||
)
|
||
data = json.loads(content)
|
||
|
||
except json.JSONDecodeError as exc:
|
||
raise ValueError(
|
||
f"Invalid JSON in {path}: {exc}"
|
||
) from exc
|
||
|
||
if not isinstance(data, dict):
|
||
raise ValueError(
|
||
"The segmentation file must contain a JSON object."
|
||
)
|
||
|
||
return data
|
||
|
||
|
||
def get_target_block_chars(
|
||
segmentation: dict[str, Any],
|
||
) -> int:
|
||
source = segmentation.get("source")
|
||
|
||
if not isinstance(source, dict):
|
||
raise ValueError(
|
||
"The segmentation JSON contains no valid source object."
|
||
)
|
||
|
||
target_chars = source.get("target_block_chars")
|
||
|
||
if not isinstance(target_chars, int):
|
||
raise ValueError(
|
||
"The segmentation JSON contains no valid "
|
||
"source.target_block_chars value."
|
||
)
|
||
|
||
if target_chars < 1:
|
||
raise ValueError(
|
||
"source.target_block_chars must be greater than zero."
|
||
)
|
||
|
||
return target_chars
|
||
|
||
|
||
def get_expected_block_count(
|
||
segmentation: dict[str, Any],
|
||
) -> int:
|
||
source = segmentation.get("source")
|
||
|
||
if not isinstance(source, dict):
|
||
raise ValueError(
|
||
"The segmentation JSON contains no valid source object."
|
||
)
|
||
|
||
block_count = source.get("block_count")
|
||
|
||
if not isinstance(block_count, int):
|
||
raise ValueError(
|
||
"The segmentation JSON contains no valid "
|
||
"source.block_count value."
|
||
)
|
||
|
||
if block_count < 1:
|
||
raise ValueError(
|
||
"source.block_count must be greater than zero."
|
||
)
|
||
|
||
return block_count
|
||
|
||
|
||
def get_segments(
|
||
segmentation: dict[str, Any],
|
||
) -> list[dict[str, Any]]:
|
||
raw_segments = segmentation.get("segments")
|
||
|
||
if not isinstance(raw_segments, list):
|
||
raise ValueError(
|
||
"The segmentation JSON contains no segments list."
|
||
)
|
||
|
||
segments: list[dict[str, Any]] = []
|
||
|
||
for index, segment in enumerate(
|
||
raw_segments,
|
||
start=1,
|
||
):
|
||
if not isinstance(segment, dict):
|
||
raise ValueError(
|
||
f"Segment {index} is not a JSON object."
|
||
)
|
||
|
||
segment_id = segment.get("segment_id")
|
||
start_block = segment.get("start_block")
|
||
end_block = segment.get("end_block")
|
||
|
||
if not isinstance(segment_id, str):
|
||
raise ValueError(
|
||
f"Segment {index} has no valid segment_id."
|
||
)
|
||
|
||
if not isinstance(start_block, int):
|
||
raise ValueError(
|
||
f"Segment {segment_id} has no valid start_block."
|
||
)
|
||
|
||
if not isinstance(end_block, int):
|
||
raise ValueError(
|
||
f"Segment {segment_id} has no valid end_block."
|
||
)
|
||
|
||
if start_block < 1:
|
||
raise ValueError(
|
||
f"Segment {segment_id} starts before block 1."
|
||
)
|
||
|
||
if end_block < start_block:
|
||
raise ValueError(
|
||
f"Segment {segment_id} ends before it starts."
|
||
)
|
||
|
||
segments.append(
|
||
{
|
||
"segment_id": segment_id,
|
||
"start_block": start_block,
|
||
"end_block": end_block,
|
||
}
|
||
)
|
||
|
||
return segments
|
||
|
||
|
||
def validate_segments(
|
||
segments: list[dict[str, Any]],
|
||
block_count: int,
|
||
) -> None:
|
||
if not segments:
|
||
raise ValueError(
|
||
"The segmentation result contains no segments."
|
||
)
|
||
|
||
expected_start = 1
|
||
|
||
for segment in segments:
|
||
start_block = segment["start_block"]
|
||
end_block = segment["end_block"]
|
||
segment_id = segment["segment_id"]
|
||
|
||
if start_block != expected_start:
|
||
raise ValueError(
|
||
f"Segment {segment_id} starts at block "
|
||
f"{start_block}; expected block {expected_start}."
|
||
)
|
||
|
||
if end_block > block_count:
|
||
raise ValueError(
|
||
f"Segment {segment_id} ends after the final block."
|
||
)
|
||
|
||
expected_start = end_block + 1
|
||
|
||
if expected_start != block_count + 1:
|
||
raise ValueError(
|
||
"The segments do not cover all transcript blocks."
|
||
)
|
||
|
||
|
||
def render_block(
|
||
block_number: int,
|
||
block_text: str,
|
||
) -> str:
|
||
return (
|
||
f"### Block {block_number}\n\n"
|
||
f"{block_text}\n"
|
||
)
|
||
|
||
|
||
def render_segment(
|
||
segment: dict[str, Any],
|
||
blocks: list[str],
|
||
) -> str:
|
||
segment_id = segment["segment_id"]
|
||
start_block = segment["start_block"]
|
||
end_block = segment["end_block"]
|
||
block_span = end_block - start_block + 1
|
||
|
||
lines = [
|
||
f"## {segment_id}",
|
||
"",
|
||
f"**Blöcke:** {start_block}–{end_block}",
|
||
"",
|
||
f"**Umfang:** {block_span} Blöcke",
|
||
"",
|
||
]
|
||
|
||
for block_number in range(
|
||
start_block,
|
||
end_block + 1,
|
||
):
|
||
lines.append(
|
||
render_block(
|
||
block_number=block_number,
|
||
block_text=blocks[block_number - 1],
|
||
)
|
||
)
|
||
|
||
return "\n".join(lines)
|
||
|
||
|
||
def render_boundary_review(
|
||
boundary: int,
|
||
blocks: list[str],
|
||
context_blocks: int,
|
||
) -> str:
|
||
block_count = len(blocks)
|
||
|
||
before_start = max(
|
||
1,
|
||
boundary - context_blocks + 1,
|
||
)
|
||
after_end = min(
|
||
block_count,
|
||
boundary + context_blocks,
|
||
)
|
||
|
||
lines = [
|
||
f"## Themenwechsel nach Block {boundary}",
|
||
"",
|
||
f"Der nächste Abschnitt beginnt mit Block {boundary + 1}.",
|
||
"",
|
||
"### Vor der Grenze",
|
||
"",
|
||
]
|
||
|
||
for block_number in range(
|
||
before_start,
|
||
boundary + 1,
|
||
):
|
||
lines.append(
|
||
render_block(
|
||
block_number=block_number,
|
||
block_text=blocks[block_number - 1],
|
||
)
|
||
)
|
||
|
||
lines.extend(
|
||
[
|
||
"",
|
||
"---",
|
||
"",
|
||
"## ⟶ Erkannter Themenwechsel",
|
||
"",
|
||
"---",
|
||
"",
|
||
"### Nach der Grenze",
|
||
"",
|
||
]
|
||
)
|
||
|
||
for block_number in range(
|
||
boundary + 1,
|
||
after_end + 1,
|
||
):
|
||
lines.append(
|
||
render_block(
|
||
block_number=block_number,
|
||
block_text=blocks[block_number - 1],
|
||
)
|
||
)
|
||
|
||
lines.extend(
|
||
[
|
||
"",
|
||
"### Manuelle Bewertung",
|
||
"",
|
||
"- [ ] echter Themenwechsel",
|
||
"- [ ] nur Unterthema",
|
||
"- [ ] kurze Abschweifung",
|
||
"- [ ] kein Themenwechsel",
|
||
"",
|
||
"**Notiz:**",
|
||
"",
|
||
"",
|
||
]
|
||
)
|
||
|
||
return "\n".join(lines)
|
||
|
||
|
||
def build_report(
|
||
transcript_file: Path,
|
||
segmentation_file: Path,
|
||
segmentation: dict[str, Any],
|
||
blocks: list[str],
|
||
segments: list[dict[str, Any]],
|
||
context_blocks: int,
|
||
) -> str:
|
||
configuration = segmentation.get(
|
||
"configuration",
|
||
{},
|
||
)
|
||
|
||
if not isinstance(configuration, dict):
|
||
configuration = {}
|
||
|
||
model = configuration.get(
|
||
"model",
|
||
"unbekannt",
|
||
)
|
||
|
||
window_size = configuration.get(
|
||
"window_size",
|
||
"unbekannt",
|
||
)
|
||
|
||
window_overlap = configuration.get(
|
||
"window_overlap",
|
||
"unbekannt",
|
||
)
|
||
|
||
boundaries = [
|
||
segment["end_block"]
|
||
for segment in segments[:-1]
|
||
]
|
||
|
||
lines = [
|
||
"# Review der Meeting-Segmentierung",
|
||
"",
|
||
"## Metadaten",
|
||
"",
|
||
f"- **Transkript:** `{transcript_file}`",
|
||
f"- **Segmentierung:** `{segmentation_file}`",
|
||
f"- **Modell:** `{model}`",
|
||
f"- **Blöcke:** {len(blocks)}",
|
||
f"- **Segmente:** {len(segments)}",
|
||
f"- **Themengrenzen:** {len(boundaries)}",
|
||
f"- **Fenstergröße:** {window_size}",
|
||
f"- **Überlappung:** {window_overlap}",
|
||
"",
|
||
"## Erkannte Grenzen",
|
||
"",
|
||
(
|
||
", ".join(str(value) for value in boundaries)
|
||
if boundaries
|
||
else "Keine"
|
||
),
|
||
"",
|
||
"---",
|
||
"",
|
||
"# Grenzprüfung",
|
||
"",
|
||
]
|
||
|
||
for boundary in boundaries:
|
||
lines.append(
|
||
render_boundary_review(
|
||
boundary=boundary,
|
||
blocks=blocks,
|
||
context_blocks=context_blocks,
|
||
)
|
||
)
|
||
|
||
lines.extend(
|
||
[
|
||
"",
|
||
"---",
|
||
"",
|
||
]
|
||
)
|
||
|
||
lines.extend(
|
||
[
|
||
"# Vollständige Segmente",
|
||
"",
|
||
]
|
||
)
|
||
|
||
for segment in segments:
|
||
lines.append(
|
||
render_segment(
|
||
segment=segment,
|
||
blocks=blocks,
|
||
)
|
||
)
|
||
|
||
lines.extend(
|
||
[
|
||
"",
|
||
"---",
|
||
"",
|
||
]
|
||
)
|
||
|
||
return "\n".join(lines).rstrip() + "\n"
|
||
|
||
|
||
def main() -> int:
|
||
args = parse_args()
|
||
|
||
try:
|
||
if not args.transcript_file.is_file():
|
||
raise FileNotFoundError(
|
||
f"Transcript file not found: "
|
||
f"{args.transcript_file}"
|
||
)
|
||
|
||
if not args.segmentation_file.is_file():
|
||
raise FileNotFoundError(
|
||
f"Segmentation file not found: "
|
||
f"{args.segmentation_file}"
|
||
)
|
||
|
||
if args.context_blocks < 1:
|
||
raise ValueError(
|
||
"context_blocks must be at least 1."
|
||
)
|
||
|
||
transcript = args.transcript_file.read_text(
|
||
encoding="utf-8-sig"
|
||
).strip()
|
||
|
||
if not transcript:
|
||
raise ValueError(
|
||
"The transcript file is empty."
|
||
)
|
||
|
||
segmentation = load_json_file(
|
||
args.segmentation_file
|
||
)
|
||
|
||
target_block_chars = get_target_block_chars(
|
||
segmentation
|
||
)
|
||
|
||
expected_block_count = get_expected_block_count(
|
||
segmentation
|
||
)
|
||
|
||
blocks = split_into_blocks(
|
||
text=transcript,
|
||
target_chars=target_block_chars,
|
||
)
|
||
|
||
if len(blocks) != expected_block_count:
|
||
raise ValueError(
|
||
"The recreated block count does not match the "
|
||
"segmentation result: "
|
||
f"{len(blocks)} instead of "
|
||
f"{expected_block_count}."
|
||
)
|
||
|
||
segments = get_segments(
|
||
segmentation
|
||
)
|
||
|
||
validate_segments(
|
||
segments=segments,
|
||
block_count=len(blocks),
|
||
)
|
||
|
||
output_path = (
|
||
args.output
|
||
or args.segmentation_file.with_name(
|
||
f"{args.segmentation_file.stem}_review.md"
|
||
)
|
||
)
|
||
|
||
output_path.parent.mkdir(
|
||
parents=True,
|
||
exist_ok=True,
|
||
)
|
||
|
||
report = build_report(
|
||
transcript_file=args.transcript_file,
|
||
segmentation_file=args.segmentation_file,
|
||
segmentation=segmentation,
|
||
blocks=blocks,
|
||
segments=segments,
|
||
context_blocks=args.context_blocks,
|
||
)
|
||
|
||
output_path.write_text(
|
||
report,
|
||
encoding="utf-8",
|
||
)
|
||
|
||
print(f"Transcript: {args.transcript_file}")
|
||
print(f"Segments: {args.segmentation_file}")
|
||
print(f"Blocks: {len(blocks)}")
|
||
print(f"Boundaries: {len(segments) - 1}")
|
||
print(f"Output: {output_path}")
|
||
|
||
return 0
|
||
|
||
except (
|
||
OSError,
|
||
UnicodeError,
|
||
ValueError,
|
||
) as exc:
|
||
print(
|
||
f"Error: {exc}",
|
||
file=sys.stderr,
|
||
)
|
||
return 1
|
||
|
||
|
||
if __name__ == "__main__":
|
||
raise SystemExit(main())
|