#!/usr/bin/env python3 """ Create a human-readable review report for transcript segmentation results. The tool reads: 1. a normalized transcript text file 2. a segmentation JSON file It recreates the analysis blocks using the same block-size setting stored in the segmentation result and writes a Markdown report containing: - all generated segments - the complete text of every segment - a compact review section around every detected boundary Example: python src/meeting_lab/segmentation/review_segmentation.py \ samples/chunks/chunk_01_normalized.txt \ samples/chunks/chunk_01_normalized_windowed_segments.json """ from __future__ import annotations import argparse import json import re import sys from pathlib import Path from typing import Any DEFAULT_CONTEXT_BLOCKS = 2 def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser( description=( "Create a readable review report for transcript segmentation." ) ) parser.add_argument( "transcript_file", type=Path, help="Normalized transcript text file", ) parser.add_argument( "segmentation_file", type=Path, help="Segmentation result JSON file", ) parser.add_argument( "-o", "--output", type=Path, help=( "Output Markdown file; default: " "_review.md" ), ) parser.add_argument( "--context-blocks", type=int, default=DEFAULT_CONTEXT_BLOCKS, help=( "Number of blocks shown before and after each boundary " f"(default: {DEFAULT_CONTEXT_BLOCKS})" ), ) return parser.parse_args() def split_into_blocks( text: str, target_chars: int, ) -> list[str]: """ Recreate the same analysis blocks used by the segmentation tool. """ if target_chars < 1: raise ValueError( "target_chars must be greater than zero." ) normalized = ( text.replace("\r\n", "\n") .replace("\r", "\n") .strip() ) units = [ part.strip() for part in re.split(r"\n\s*\n+", normalized) if part.strip() ] if not units: units = [ line.strip() for line in normalized.splitlines() if line.strip() ] blocks: list[str] = [] current_units: list[str] = [] current_length = 0 for unit in units: separator_length = 1 if current_units else 0 projected_length = ( current_length + separator_length + len(unit) ) if current_units and projected_length > target_chars: blocks.append( "\n".join(current_units) ) current_units = [unit] current_length = len(unit) else: current_units.append(unit) current_length = projected_length if current_units: blocks.append( "\n".join(current_units) ) return blocks def load_json_file( path: Path, ) -> dict[str, Any]: try: content = path.read_text( encoding="utf-8-sig" ) data = json.loads(content) except json.JSONDecodeError as exc: raise ValueError( f"Invalid JSON in {path}: {exc}" ) from exc if not isinstance(data, dict): raise ValueError( "The segmentation file must contain a JSON object." ) return data def get_target_block_chars( segmentation: dict[str, Any], ) -> int: source = segmentation.get("source") if not isinstance(source, dict): raise ValueError( "The segmentation JSON contains no valid source object." ) target_chars = source.get("target_block_chars") if not isinstance(target_chars, int): raise ValueError( "The segmentation JSON contains no valid " "source.target_block_chars value." ) if target_chars < 1: raise ValueError( "source.target_block_chars must be greater than zero." ) return target_chars def get_expected_block_count( segmentation: dict[str, Any], ) -> int: source = segmentation.get("source") if not isinstance(source, dict): raise ValueError( "The segmentation JSON contains no valid source object." ) block_count = source.get("block_count") if not isinstance(block_count, int): raise ValueError( "The segmentation JSON contains no valid " "source.block_count value." ) if block_count < 1: raise ValueError( "source.block_count must be greater than zero." ) return block_count def get_segments( segmentation: dict[str, Any], ) -> list[dict[str, Any]]: raw_segments = segmentation.get("segments") if not isinstance(raw_segments, list): raise ValueError( "The segmentation JSON contains no segments list." ) segments: list[dict[str, Any]] = [] for index, segment in enumerate( raw_segments, start=1, ): if not isinstance(segment, dict): raise ValueError( f"Segment {index} is not a JSON object." ) segment_id = segment.get("segment_id") start_block = segment.get("start_block") end_block = segment.get("end_block") if not isinstance(segment_id, str): raise ValueError( f"Segment {index} has no valid segment_id." ) if not isinstance(start_block, int): raise ValueError( f"Segment {segment_id} has no valid start_block." ) if not isinstance(end_block, int): raise ValueError( f"Segment {segment_id} has no valid end_block." ) if start_block < 1: raise ValueError( f"Segment {segment_id} starts before block 1." ) if end_block < start_block: raise ValueError( f"Segment {segment_id} ends before it starts." ) segments.append( { "segment_id": segment_id, "start_block": start_block, "end_block": end_block, } ) return segments def validate_segments( segments: list[dict[str, Any]], block_count: int, ) -> None: if not segments: raise ValueError( "The segmentation result contains no segments." ) expected_start = 1 for segment in segments: start_block = segment["start_block"] end_block = segment["end_block"] segment_id = segment["segment_id"] if start_block != expected_start: raise ValueError( f"Segment {segment_id} starts at block " f"{start_block}; expected block {expected_start}." ) if end_block > block_count: raise ValueError( f"Segment {segment_id} ends after the final block." ) expected_start = end_block + 1 if expected_start != block_count + 1: raise ValueError( "The segments do not cover all transcript blocks." ) def render_block( block_number: int, block_text: str, ) -> str: return ( f"### Block {block_number}\n\n" f"{block_text}\n" ) def render_segment( segment: dict[str, Any], blocks: list[str], ) -> str: segment_id = segment["segment_id"] start_block = segment["start_block"] end_block = segment["end_block"] block_span = end_block - start_block + 1 lines = [ f"## {segment_id}", "", f"**Blöcke:** {start_block}–{end_block}", "", f"**Umfang:** {block_span} Blöcke", "", ] for block_number in range( start_block, end_block + 1, ): lines.append( render_block( block_number=block_number, block_text=blocks[block_number - 1], ) ) return "\n".join(lines) def render_boundary_review( boundary: int, blocks: list[str], context_blocks: int, ) -> str: block_count = len(blocks) before_start = max( 1, boundary - context_blocks + 1, ) after_end = min( block_count, boundary + context_blocks, ) lines = [ f"## Themenwechsel nach Block {boundary}", "", f"Der nächste Abschnitt beginnt mit Block {boundary + 1}.", "", "### Vor der Grenze", "", ] for block_number in range( before_start, boundary + 1, ): lines.append( render_block( block_number=block_number, block_text=blocks[block_number - 1], ) ) lines.extend( [ "", "---", "", "## ⟶ Erkannter Themenwechsel", "", "---", "", "### Nach der Grenze", "", ] ) for block_number in range( boundary + 1, after_end + 1, ): lines.append( render_block( block_number=block_number, block_text=blocks[block_number - 1], ) ) lines.extend( [ "", "### Manuelle Bewertung", "", "- [ ] echter Themenwechsel", "- [ ] nur Unterthema", "- [ ] kurze Abschweifung", "- [ ] kein Themenwechsel", "", "**Notiz:**", "", "", ] ) return "\n".join(lines) def build_report( transcript_file: Path, segmentation_file: Path, segmentation: dict[str, Any], blocks: list[str], segments: list[dict[str, Any]], context_blocks: int, ) -> str: configuration = segmentation.get( "configuration", {}, ) if not isinstance(configuration, dict): configuration = {} model = configuration.get( "model", "unbekannt", ) window_size = configuration.get( "window_size", "unbekannt", ) window_overlap = configuration.get( "window_overlap", "unbekannt", ) boundaries = [ segment["end_block"] for segment in segments[:-1] ] lines = [ "# Review der Meeting-Segmentierung", "", "## Metadaten", "", f"- **Transkript:** `{transcript_file}`", f"- **Segmentierung:** `{segmentation_file}`", f"- **Modell:** `{model}`", f"- **Blöcke:** {len(blocks)}", f"- **Segmente:** {len(segments)}", f"- **Themengrenzen:** {len(boundaries)}", f"- **Fenstergröße:** {window_size}", f"- **Überlappung:** {window_overlap}", "", "## Erkannte Grenzen", "", ( ", ".join(str(value) for value in boundaries) if boundaries else "Keine" ), "", "---", "", "# Grenzprüfung", "", ] for boundary in boundaries: lines.append( render_boundary_review( boundary=boundary, blocks=blocks, context_blocks=context_blocks, ) ) lines.extend( [ "", "---", "", ] ) lines.extend( [ "# Vollständige Segmente", "", ] ) for segment in segments: lines.append( render_segment( segment=segment, blocks=blocks, ) ) lines.extend( [ "", "---", "", ] ) return "\n".join(lines).rstrip() + "\n" def main() -> int: args = parse_args() try: if not args.transcript_file.is_file(): raise FileNotFoundError( f"Transcript file not found: " f"{args.transcript_file}" ) if not args.segmentation_file.is_file(): raise FileNotFoundError( f"Segmentation file not found: " f"{args.segmentation_file}" ) if args.context_blocks < 1: raise ValueError( "context_blocks must be at least 1." ) transcript = args.transcript_file.read_text( encoding="utf-8-sig" ).strip() if not transcript: raise ValueError( "The transcript file is empty." ) segmentation = load_json_file( args.segmentation_file ) target_block_chars = get_target_block_chars( segmentation ) expected_block_count = get_expected_block_count( segmentation ) blocks = split_into_blocks( text=transcript, target_chars=target_block_chars, ) if len(blocks) != expected_block_count: raise ValueError( "The recreated block count does not match the " "segmentation result: " f"{len(blocks)} instead of " f"{expected_block_count}." ) segments = get_segments( segmentation ) validate_segments( segments=segments, block_count=len(blocks), ) output_path = ( args.output or args.segmentation_file.with_name( f"{args.segmentation_file.stem}_review.md" ) ) output_path.parent.mkdir( parents=True, exist_ok=True, ) report = build_report( transcript_file=args.transcript_file, segmentation_file=args.segmentation_file, segmentation=segmentation, blocks=blocks, segments=segments, context_blocks=args.context_blocks, ) output_path.write_text( report, encoding="utf-8", ) print(f"Transcript: {args.transcript_file}") print(f"Segments: {args.segmentation_file}") print(f"Blocks: {len(blocks)}") print(f"Boundaries: {len(segments) - 1}") print(f"Output: {output_path}") return 0 except ( OSError, UnicodeError, ValueError, ) as exc: print( f"Error: {exc}", file=sys.stderr, ) return 1 if __name__ == "__main__": raise SystemExit(main())