Files
meeting-lab/src/meeting_lab/segmentation/review_segmentation.py
T

654 lines
15 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
Create a human-readable review report for transcript segmentation results.
The tool reads:
1. a normalized transcript text file
2. a segmentation JSON file
It recreates the analysis blocks using the same block-size setting stored in
the segmentation result and writes a Markdown report containing:
- all generated segments
- the complete text of every segment
- a compact review section around every detected boundary
Example:
python src/meeting_lab/segmentation/review_segmentation.py \
samples/chunks/chunk_01_normalized.txt \
samples/chunks/chunk_01_normalized_windowed_segments.json
"""
from __future__ import annotations
import argparse
import json
import re
import sys
from pathlib import Path
from typing import Any
DEFAULT_CONTEXT_BLOCKS = 2
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description=(
"Create a readable review report for transcript segmentation."
)
)
parser.add_argument(
"transcript_file",
type=Path,
help="Normalized transcript text file",
)
parser.add_argument(
"segmentation_file",
type=Path,
help="Segmentation result JSON file",
)
parser.add_argument(
"-o",
"--output",
type=Path,
help=(
"Output Markdown file; default: "
"<segmentation_file>_review.md"
),
)
parser.add_argument(
"--context-blocks",
type=int,
default=DEFAULT_CONTEXT_BLOCKS,
help=(
"Number of blocks shown before and after each boundary "
f"(default: {DEFAULT_CONTEXT_BLOCKS})"
),
)
return parser.parse_args()
def split_into_blocks(
text: str,
target_chars: int,
) -> list[str]:
"""
Recreate the same analysis blocks used by the segmentation tool.
"""
if target_chars < 1:
raise ValueError(
"target_chars must be greater than zero."
)
normalized = (
text.replace("\r\n", "\n")
.replace("\r", "\n")
.strip()
)
units = [
part.strip()
for part in re.split(r"\n\s*\n+", normalized)
if part.strip()
]
if not units:
units = [
line.strip()
for line in normalized.splitlines()
if line.strip()
]
blocks: list[str] = []
current_units: list[str] = []
current_length = 0
for unit in units:
separator_length = 1 if current_units else 0
projected_length = (
current_length
+ separator_length
+ len(unit)
)
if current_units and projected_length > target_chars:
blocks.append(
"\n".join(current_units)
)
current_units = [unit]
current_length = len(unit)
else:
current_units.append(unit)
current_length = projected_length
if current_units:
blocks.append(
"\n".join(current_units)
)
return blocks
def load_json_file(
path: Path,
) -> dict[str, Any]:
try:
content = path.read_text(
encoding="utf-8-sig"
)
data = json.loads(content)
except json.JSONDecodeError as exc:
raise ValueError(
f"Invalid JSON in {path}: {exc}"
) from exc
if not isinstance(data, dict):
raise ValueError(
"The segmentation file must contain a JSON object."
)
return data
def get_target_block_chars(
segmentation: dict[str, Any],
) -> int:
source = segmentation.get("source")
if not isinstance(source, dict):
raise ValueError(
"The segmentation JSON contains no valid source object."
)
target_chars = source.get("target_block_chars")
if not isinstance(target_chars, int):
raise ValueError(
"The segmentation JSON contains no valid "
"source.target_block_chars value."
)
if target_chars < 1:
raise ValueError(
"source.target_block_chars must be greater than zero."
)
return target_chars
def get_expected_block_count(
segmentation: dict[str, Any],
) -> int:
source = segmentation.get("source")
if not isinstance(source, dict):
raise ValueError(
"The segmentation JSON contains no valid source object."
)
block_count = source.get("block_count")
if not isinstance(block_count, int):
raise ValueError(
"The segmentation JSON contains no valid "
"source.block_count value."
)
if block_count < 1:
raise ValueError(
"source.block_count must be greater than zero."
)
return block_count
def get_segments(
segmentation: dict[str, Any],
) -> list[dict[str, Any]]:
raw_segments = segmentation.get("segments")
if not isinstance(raw_segments, list):
raise ValueError(
"The segmentation JSON contains no segments list."
)
segments: list[dict[str, Any]] = []
for index, segment in enumerate(
raw_segments,
start=1,
):
if not isinstance(segment, dict):
raise ValueError(
f"Segment {index} is not a JSON object."
)
segment_id = segment.get("segment_id")
start_block = segment.get("start_block")
end_block = segment.get("end_block")
if not isinstance(segment_id, str):
raise ValueError(
f"Segment {index} has no valid segment_id."
)
if not isinstance(start_block, int):
raise ValueError(
f"Segment {segment_id} has no valid start_block."
)
if not isinstance(end_block, int):
raise ValueError(
f"Segment {segment_id} has no valid end_block."
)
if start_block < 1:
raise ValueError(
f"Segment {segment_id} starts before block 1."
)
if end_block < start_block:
raise ValueError(
f"Segment {segment_id} ends before it starts."
)
segments.append(
{
"segment_id": segment_id,
"start_block": start_block,
"end_block": end_block,
}
)
return segments
def validate_segments(
segments: list[dict[str, Any]],
block_count: int,
) -> None:
if not segments:
raise ValueError(
"The segmentation result contains no segments."
)
expected_start = 1
for segment in segments:
start_block = segment["start_block"]
end_block = segment["end_block"]
segment_id = segment["segment_id"]
if start_block != expected_start:
raise ValueError(
f"Segment {segment_id} starts at block "
f"{start_block}; expected block {expected_start}."
)
if end_block > block_count:
raise ValueError(
f"Segment {segment_id} ends after the final block."
)
expected_start = end_block + 1
if expected_start != block_count + 1:
raise ValueError(
"The segments do not cover all transcript blocks."
)
def render_block(
block_number: int,
block_text: str,
) -> str:
return (
f"### Block {block_number}\n\n"
f"{block_text}\n"
)
def render_segment(
segment: dict[str, Any],
blocks: list[str],
) -> str:
segment_id = segment["segment_id"]
start_block = segment["start_block"]
end_block = segment["end_block"]
block_span = end_block - start_block + 1
lines = [
f"## {segment_id}",
"",
f"**Blöcke:** {start_block}–{end_block}",
"",
f"**Umfang:** {block_span} Blöcke",
"",
]
for block_number in range(
start_block,
end_block + 1,
):
lines.append(
render_block(
block_number=block_number,
block_text=blocks[block_number - 1],
)
)
return "\n".join(lines)
def render_boundary_review(
boundary: int,
blocks: list[str],
context_blocks: int,
) -> str:
block_count = len(blocks)
before_start = max(
1,
boundary - context_blocks + 1,
)
after_end = min(
block_count,
boundary + context_blocks,
)
lines = [
f"## Themenwechsel nach Block {boundary}",
"",
f"Der nächste Abschnitt beginnt mit Block {boundary + 1}.",
"",
"### Vor der Grenze",
"",
]
for block_number in range(
before_start,
boundary + 1,
):
lines.append(
render_block(
block_number=block_number,
block_text=blocks[block_number - 1],
)
)
lines.extend(
[
"",
"---",
"",
"## ⟶ Erkannter Themenwechsel",
"",
"---",
"",
"### Nach der Grenze",
"",
]
)
for block_number in range(
boundary + 1,
after_end + 1,
):
lines.append(
render_block(
block_number=block_number,
block_text=blocks[block_number - 1],
)
)
lines.extend(
[
"",
"### Manuelle Bewertung",
"",
"- [ ] echter Themenwechsel",
"- [ ] nur Unterthema",
"- [ ] kurze Abschweifung",
"- [ ] kein Themenwechsel",
"",
"**Notiz:**",
"",
"",
]
)
return "\n".join(lines)
def build_report(
transcript_file: Path,
segmentation_file: Path,
segmentation: dict[str, Any],
blocks: list[str],
segments: list[dict[str, Any]],
context_blocks: int,
) -> str:
configuration = segmentation.get(
"configuration",
{},
)
if not isinstance(configuration, dict):
configuration = {}
model = configuration.get(
"model",
"unbekannt",
)
window_size = configuration.get(
"window_size",
"unbekannt",
)
window_overlap = configuration.get(
"window_overlap",
"unbekannt",
)
boundaries = [
segment["end_block"]
for segment in segments[:-1]
]
lines = [
"# Review der Meeting-Segmentierung",
"",
"## Metadaten",
"",
f"- **Transkript:** `{transcript_file}`",
f"- **Segmentierung:** `{segmentation_file}`",
f"- **Modell:** `{model}`",
f"- **Blöcke:** {len(blocks)}",
f"- **Segmente:** {len(segments)}",
f"- **Themengrenzen:** {len(boundaries)}",
f"- **Fenstergröße:** {window_size}",
f"- **Überlappung:** {window_overlap}",
"",
"## Erkannte Grenzen",
"",
(
", ".join(str(value) for value in boundaries)
if boundaries
else "Keine"
),
"",
"---",
"",
"# Grenzprüfung",
"",
]
for boundary in boundaries:
lines.append(
render_boundary_review(
boundary=boundary,
blocks=blocks,
context_blocks=context_blocks,
)
)
lines.extend(
[
"",
"---",
"",
]
)
lines.extend(
[
"# Vollständige Segmente",
"",
]
)
for segment in segments:
lines.append(
render_segment(
segment=segment,
blocks=blocks,
)
)
lines.extend(
[
"",
"---",
"",
]
)
return "\n".join(lines).rstrip() + "\n"
def main() -> int:
args = parse_args()
try:
if not args.transcript_file.is_file():
raise FileNotFoundError(
f"Transcript file not found: "
f"{args.transcript_file}"
)
if not args.segmentation_file.is_file():
raise FileNotFoundError(
f"Segmentation file not found: "
f"{args.segmentation_file}"
)
if args.context_blocks < 1:
raise ValueError(
"context_blocks must be at least 1."
)
transcript = args.transcript_file.read_text(
encoding="utf-8-sig"
).strip()
if not transcript:
raise ValueError(
"The transcript file is empty."
)
segmentation = load_json_file(
args.segmentation_file
)
target_block_chars = get_target_block_chars(
segmentation
)
expected_block_count = get_expected_block_count(
segmentation
)
blocks = split_into_blocks(
text=transcript,
target_chars=target_block_chars,
)
if len(blocks) != expected_block_count:
raise ValueError(
"The recreated block count does not match the "
"segmentation result: "
f"{len(blocks)} instead of "
f"{expected_block_count}."
)
segments = get_segments(
segmentation
)
validate_segments(
segments=segments,
block_count=len(blocks),
)
output_path = (
args.output
or args.segmentation_file.with_name(
f"{args.segmentation_file.stem}_review.md"
)
)
output_path.parent.mkdir(
parents=True,
exist_ok=True,
)
report = build_report(
transcript_file=args.transcript_file,
segmentation_file=args.segmentation_file,
segmentation=segmentation,
blocks=blocks,
segments=segments,
context_blocks=args.context_blocks,
)
output_path.write_text(
report,
encoding="utf-8",
)
print(f"Transcript: {args.transcript_file}")
print(f"Segments: {args.segmentation_file}")
print(f"Blocks: {len(blocks)}")
print(f"Boundaries: {len(segments) - 1}")
print(f"Output: {output_path}")
return 0
except (
OSError,
UnicodeError,
ValueError,
) as exc:
print(
f"Error: {exc}",
file=sys.stderr,
)
return 1
if __name__ == "__main__":
raise SystemExit(main())