Add whisper.cpp JSON converter
This commit is contained in:
@@ -0,0 +1,210 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Convert whisper.cpp JSON output to the compact Meeting Lab transcript shape."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
class WhisperCppConversionError(ValueError):
|
||||
"""Raised when whisper.cpp JSON cannot be converted deterministically."""
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Convert whisper.cpp JSON output to compact JSON and text files."
|
||||
)
|
||||
parser.add_argument("input_json", type=Path, help="whisper.cpp JSON file")
|
||||
parser.add_argument(
|
||||
"--json-output",
|
||||
type=Path,
|
||||
help="Compact JSON output path; default: <input>_converted.json",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--text-output",
|
||||
type=Path,
|
||||
help="Readable text output path; default: <input>_converted.txt",
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def load_json(path: Path) -> dict[str, Any]:
|
||||
data = json.loads(path.read_text(encoding="utf-8"))
|
||||
if not isinstance(data, dict):
|
||||
raise WhisperCppConversionError("Input JSON must contain a top-level object.")
|
||||
return data
|
||||
|
||||
|
||||
def transcription_entries(data: dict[str, Any]) -> list[Any]:
|
||||
entries = data.get("transcription")
|
||||
if not isinstance(entries, list):
|
||||
raise WhisperCppConversionError(
|
||||
"Input JSON must contain a 'transcription' list."
|
||||
)
|
||||
return entries
|
||||
|
||||
|
||||
def offset_seconds(entry: dict[str, Any], entry_index: int, field: str) -> float:
|
||||
offsets = entry.get("offsets")
|
||||
if not isinstance(offsets, dict):
|
||||
raise WhisperCppConversionError(
|
||||
f"transcription[{entry_index}].offsets must be an object."
|
||||
)
|
||||
value = offsets.get(field)
|
||||
if not isinstance(value, (int, float)):
|
||||
raise WhisperCppConversionError(
|
||||
f"transcription[{entry_index}].offsets.{field} must be a number."
|
||||
)
|
||||
return float(value) / 1000.0
|
||||
|
||||
|
||||
def validate_timestamp_fields(entry: dict[str, Any], entry_index: int) -> None:
|
||||
timestamps = entry.get("timestamps")
|
||||
if not isinstance(timestamps, dict):
|
||||
raise WhisperCppConversionError(
|
||||
f"transcription[{entry_index}].timestamps must be an object."
|
||||
)
|
||||
for field in ("from", "to"):
|
||||
if field not in timestamps:
|
||||
raise WhisperCppConversionError(
|
||||
f"transcription[{entry_index}].timestamps.{field} is missing."
|
||||
)
|
||||
|
||||
|
||||
def convert_entries(entries: list[Any]) -> dict[str, Any]:
|
||||
segments: list[dict[str, Any]] = []
|
||||
|
||||
for entry_index, entry in enumerate(entries):
|
||||
if not isinstance(entry, dict):
|
||||
raise WhisperCppConversionError(
|
||||
f"transcription[{entry_index}] must be an object."
|
||||
)
|
||||
validate_timestamp_fields(entry, entry_index)
|
||||
start = offset_seconds(entry, entry_index, "from")
|
||||
end = offset_seconds(entry, entry_index, "to")
|
||||
if end < start:
|
||||
raise WhisperCppConversionError(
|
||||
f"transcription[{entry_index}].offsets.to must be greater than or equal to offsets.from."
|
||||
)
|
||||
|
||||
text_value = entry.get("text", "")
|
||||
if not isinstance(text_value, str):
|
||||
raise WhisperCppConversionError(
|
||||
f"transcription[{entry_index}].text must be a string."
|
||||
)
|
||||
text = text_value.strip()
|
||||
if not text:
|
||||
continue
|
||||
|
||||
segments.append(
|
||||
{
|
||||
"id": len(segments),
|
||||
"start": start,
|
||||
"end": end,
|
||||
"text": text,
|
||||
}
|
||||
)
|
||||
|
||||
return {
|
||||
"text": " ".join(segment["text"] for segment in segments),
|
||||
"segments": segments,
|
||||
}
|
||||
|
||||
|
||||
def format_timestamp(seconds: float) -> str:
|
||||
milliseconds_total = int(round(seconds * 1000))
|
||||
milliseconds = milliseconds_total % 1000
|
||||
seconds_total = milliseconds_total // 1000
|
||||
second = seconds_total % 60
|
||||
minutes_total = seconds_total // 60
|
||||
minute = minutes_total % 60
|
||||
hour = minutes_total // 60
|
||||
return f"{hour:02d}:{minute:02d}:{second:02d}.{milliseconds:03d}"
|
||||
|
||||
|
||||
def render_text(segments: list[dict[str, Any]]) -> str:
|
||||
lines = [
|
||||
(
|
||||
f"[{format_timestamp(segment['start'])} - "
|
||||
f"{format_timestamp(segment['end'])}] {segment['text']}"
|
||||
)
|
||||
for segment in segments
|
||||
]
|
||||
return "\n".join(lines) + ("\n" if lines else "")
|
||||
|
||||
|
||||
def default_json_output(input_path: Path) -> Path:
|
||||
return input_path.with_name(f"{input_path.stem}_converted.json")
|
||||
|
||||
|
||||
def default_text_output(input_path: Path) -> Path:
|
||||
return input_path.with_name(f"{input_path.stem}_converted.txt")
|
||||
|
||||
|
||||
def convert_whispercpp_json(
|
||||
input_path: Path,
|
||||
json_output_path: Path | None = None,
|
||||
text_output_path: Path | None = None,
|
||||
) -> dict[str, Any]:
|
||||
data = load_json(input_path)
|
||||
entries = transcription_entries(data)
|
||||
converted = convert_entries(entries)
|
||||
json_output = json_output_path or default_json_output(input_path)
|
||||
text_output = text_output_path or default_text_output(input_path)
|
||||
|
||||
json_output.parent.mkdir(parents=True, exist_ok=True)
|
||||
text_output.parent.mkdir(parents=True, exist_ok=True)
|
||||
json_output.write_text(
|
||||
json.dumps(converted, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
text_output.write_text(render_text(converted["segments"]), encoding="utf-8")
|
||||
|
||||
total_duration = (
|
||||
max((segment["end"] for segment in converted["segments"]), default=0.0)
|
||||
)
|
||||
summary = {
|
||||
"input_path": input_path,
|
||||
"json_output_path": json_output,
|
||||
"text_output_path": text_output,
|
||||
"original_entry_count": len(entries),
|
||||
"retained_segment_count": len(converted["segments"]),
|
||||
"total_character_count": len(converted["text"]),
|
||||
"detected_total_duration": total_duration,
|
||||
}
|
||||
print_summary(summary)
|
||||
return summary
|
||||
|
||||
|
||||
def print_summary(summary: dict[str, Any]) -> None:
|
||||
print(f"Input path: {summary['input_path']}")
|
||||
print(f"JSON output path: {summary['json_output_path']}")
|
||||
print(f"Text output path: {summary['text_output_path']}")
|
||||
print(f"Original entry count: {summary['original_entry_count']}")
|
||||
print(f"Retained segment count: {summary['retained_segment_count']}")
|
||||
print(f"Total character count: {summary['total_character_count']}")
|
||||
print(
|
||||
"Detected total duration: "
|
||||
f"{summary['detected_total_duration']:.3f} seconds"
|
||||
)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
try:
|
||||
convert_whispercpp_json(
|
||||
input_path=args.input_json,
|
||||
json_output_path=args.json_output,
|
||||
text_output_path=args.text_output,
|
||||
)
|
||||
except (OSError, UnicodeError, json.JSONDecodeError, WhisperCppConversionError) as exc:
|
||||
print(f"Error: {exc}")
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user