Add whisper.cpp JSON converter

This commit is contained in:
2026-08-04 16:57:28 +02:00
parent 647b28aa3e
commit 74aa246151
2 changed files with 355 additions and 0 deletions
+210
View File
@@ -0,0 +1,210 @@
#!/usr/bin/env python3
"""Convert whisper.cpp JSON output to the compact Meeting Lab transcript shape."""
from __future__ import annotations
import argparse
import json
from pathlib import Path
from typing import Any
class WhisperCppConversionError(ValueError):
"""Raised when whisper.cpp JSON cannot be converted deterministically."""
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Convert whisper.cpp JSON output to compact JSON and text files."
)
parser.add_argument("input_json", type=Path, help="whisper.cpp JSON file")
parser.add_argument(
"--json-output",
type=Path,
help="Compact JSON output path; default: <input>_converted.json",
)
parser.add_argument(
"--text-output",
type=Path,
help="Readable text output path; default: <input>_converted.txt",
)
return parser.parse_args()
def load_json(path: Path) -> dict[str, Any]:
data = json.loads(path.read_text(encoding="utf-8"))
if not isinstance(data, dict):
raise WhisperCppConversionError("Input JSON must contain a top-level object.")
return data
def transcription_entries(data: dict[str, Any]) -> list[Any]:
entries = data.get("transcription")
if not isinstance(entries, list):
raise WhisperCppConversionError(
"Input JSON must contain a 'transcription' list."
)
return entries
def offset_seconds(entry: dict[str, Any], entry_index: int, field: str) -> float:
offsets = entry.get("offsets")
if not isinstance(offsets, dict):
raise WhisperCppConversionError(
f"transcription[{entry_index}].offsets must be an object."
)
value = offsets.get(field)
if not isinstance(value, (int, float)):
raise WhisperCppConversionError(
f"transcription[{entry_index}].offsets.{field} must be a number."
)
return float(value) / 1000.0
def validate_timestamp_fields(entry: dict[str, Any], entry_index: int) -> None:
timestamps = entry.get("timestamps")
if not isinstance(timestamps, dict):
raise WhisperCppConversionError(
f"transcription[{entry_index}].timestamps must be an object."
)
for field in ("from", "to"):
if field not in timestamps:
raise WhisperCppConversionError(
f"transcription[{entry_index}].timestamps.{field} is missing."
)
def convert_entries(entries: list[Any]) -> dict[str, Any]:
segments: list[dict[str, Any]] = []
for entry_index, entry in enumerate(entries):
if not isinstance(entry, dict):
raise WhisperCppConversionError(
f"transcription[{entry_index}] must be an object."
)
validate_timestamp_fields(entry, entry_index)
start = offset_seconds(entry, entry_index, "from")
end = offset_seconds(entry, entry_index, "to")
if end < start:
raise WhisperCppConversionError(
f"transcription[{entry_index}].offsets.to must be greater than or equal to offsets.from."
)
text_value = entry.get("text", "")
if not isinstance(text_value, str):
raise WhisperCppConversionError(
f"transcription[{entry_index}].text must be a string."
)
text = text_value.strip()
if not text:
continue
segments.append(
{
"id": len(segments),
"start": start,
"end": end,
"text": text,
}
)
return {
"text": " ".join(segment["text"] for segment in segments),
"segments": segments,
}
def format_timestamp(seconds: float) -> str:
milliseconds_total = int(round(seconds * 1000))
milliseconds = milliseconds_total % 1000
seconds_total = milliseconds_total // 1000
second = seconds_total % 60
minutes_total = seconds_total // 60
minute = minutes_total % 60
hour = minutes_total // 60
return f"{hour:02d}:{minute:02d}:{second:02d}.{milliseconds:03d}"
def render_text(segments: list[dict[str, Any]]) -> str:
lines = [
(
f"[{format_timestamp(segment['start'])} - "
f"{format_timestamp(segment['end'])}] {segment['text']}"
)
for segment in segments
]
return "\n".join(lines) + ("\n" if lines else "")
def default_json_output(input_path: Path) -> Path:
return input_path.with_name(f"{input_path.stem}_converted.json")
def default_text_output(input_path: Path) -> Path:
return input_path.with_name(f"{input_path.stem}_converted.txt")
def convert_whispercpp_json(
input_path: Path,
json_output_path: Path | None = None,
text_output_path: Path | None = None,
) -> dict[str, Any]:
data = load_json(input_path)
entries = transcription_entries(data)
converted = convert_entries(entries)
json_output = json_output_path or default_json_output(input_path)
text_output = text_output_path or default_text_output(input_path)
json_output.parent.mkdir(parents=True, exist_ok=True)
text_output.parent.mkdir(parents=True, exist_ok=True)
json_output.write_text(
json.dumps(converted, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
text_output.write_text(render_text(converted["segments"]), encoding="utf-8")
total_duration = (
max((segment["end"] for segment in converted["segments"]), default=0.0)
)
summary = {
"input_path": input_path,
"json_output_path": json_output,
"text_output_path": text_output,
"original_entry_count": len(entries),
"retained_segment_count": len(converted["segments"]),
"total_character_count": len(converted["text"]),
"detected_total_duration": total_duration,
}
print_summary(summary)
return summary
def print_summary(summary: dict[str, Any]) -> None:
print(f"Input path: {summary['input_path']}")
print(f"JSON output path: {summary['json_output_path']}")
print(f"Text output path: {summary['text_output_path']}")
print(f"Original entry count: {summary['original_entry_count']}")
print(f"Retained segment count: {summary['retained_segment_count']}")
print(f"Total character count: {summary['total_character_count']}")
print(
"Detected total duration: "
f"{summary['detected_total_duration']:.3f} seconds"
)
def main() -> int:
args = parse_args()
try:
convert_whispercpp_json(
input_path=args.input_json,
json_output_path=args.json_output,
text_output_path=args.text_output,
)
except (OSError, UnicodeError, json.JSONDecodeError, WhisperCppConversionError) as exc:
print(f"Error: {exc}")
return 1
return 0
if __name__ == "__main__":
raise SystemExit(main())
+145
View File
@@ -0,0 +1,145 @@
import json
import tempfile
import unittest
from pathlib import Path
from scripts.convert_whispercpp_json import (
WhisperCppConversionError,
convert_entries,
convert_whispercpp_json,
format_timestamp,
)
def whispercpp_entry(
text: str,
offset_from: int,
offset_to: int,
timestamp_from: str = "00:00:00,000",
timestamp_to: str = "00:00:01,000",
) -> dict:
return {
"timestamps": {"from": timestamp_from, "to": timestamp_to},
"offsets": {"from": offset_from, "to": offset_to},
"text": text,
}
class ConvertWhisperCppJsonTests(unittest.TestCase):
def test_valid_whispercpp_json_writes_compact_json_and_text(self) -> None:
with tempfile.TemporaryDirectory() as directory:
root = Path(directory)
input_path = root / "whispercpp.json"
json_output = root / "converted.json"
text_output = root / "converted.txt"
input_path.write_text(
json.dumps(
{
"transcription": [
whispercpp_entry(" Hello ", 0, 12340),
whispercpp_entry("world", 12340, 13000),
]
},
ensure_ascii=False,
),
encoding="utf-8",
)
summary = convert_whispercpp_json(input_path, json_output, text_output)
converted = json.loads(json_output.read_text(encoding="utf-8"))
self.assertEqual(converted["text"], "Hello world")
self.assertEqual(
converted["segments"],
[
{"id": 0, "start": 0.0, "end": 12.34, "text": "Hello"},
{"id": 1, "start": 12.34, "end": 13.0, "text": "world"},
],
)
self.assertEqual(
text_output.read_text(encoding="utf-8"),
"[00:00:00.000 - 00:00:12.340] Hello\n"
"[00:00:12.340 - 00:00:13.000] world\n",
)
self.assertEqual(summary["original_entry_count"], 2)
self.assertEqual(summary["retained_segment_count"], 2)
self.assertEqual(summary["total_character_count"], len("Hello world"))
self.assertEqual(summary["detected_total_duration"], 13.0)
def test_empty_segments_are_skipped_and_ids_are_renumbered(self) -> None:
converted = convert_entries(
[
whispercpp_entry("first", 0, 1000),
whispercpp_entry(" ", 1000, 2000),
whispercpp_entry("", 2000, 3000),
whispercpp_entry("second", 3000, 4000),
]
)
self.assertEqual(converted["text"], "first second")
self.assertEqual(
converted["segments"],
[
{"id": 0, "start": 0.0, "end": 1.0, "text": "first"},
{"id": 1, "start": 3.0, "end": 4.0, "text": "second"},
],
)
def test_missing_transcription_list_is_rejected(self) -> None:
with tempfile.TemporaryDirectory() as directory:
input_path = Path(directory) / "missing.json"
input_path.write_text(json.dumps({"segments": []}), encoding="utf-8")
with self.assertRaisesRegex(
WhisperCppConversionError,
"Input JSON must contain a 'transcription' list.",
):
convert_whispercpp_json(input_path)
def test_malformed_offsets_are_rejected(self) -> None:
with self.assertRaisesRegex(
WhisperCppConversionError,
r"transcription\[0\]\.offsets\.from must be a number.",
):
convert_entries([whispercpp_entry("bad", "0", 1000)])
with self.assertRaisesRegex(
WhisperCppConversionError,
r"offsets\.to must be greater than or equal to offsets\.from",
):
convert_entries([whispercpp_entry("bad", 2000, 1000)])
def test_utf8_text_is_preserved(self) -> None:
converted = convert_entries(
[
whispercpp_entry(" Grüße für Marleen – HÜSKER ", 0, 1000),
]
)
self.assertEqual(converted["text"], "Grüße für Marleen – HÜSKER")
self.assertEqual(
converted["segments"][0]["text"],
"Grüße für Marleen – HÜSKER",
)
def test_ordering_is_preserved(self) -> None:
converted = convert_entries(
[
whispercpp_entry("third spoken first", 3000, 4000),
whispercpp_entry("first spoken second", 0, 1000),
whispercpp_entry("second spoken third", 1000, 2000),
]
)
self.assertEqual(
[segment["text"] for segment in converted["segments"]],
["third spoken first", "first spoken second", "second spoken third"],
)
def test_format_timestamp_uses_hours_minutes_seconds_milliseconds(self) -> None:
self.assertEqual(format_timestamp(123.45), "00:02:03.450")
self.assertEqual(format_timestamp(3723.004), "01:02:03.004")
if __name__ == "__main__":
unittest.main()