Compare commits
2
Commits
647b28aa3e
...
dcc3a5c734
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
dcc3a5c734 | ||
|
|
74aa246151 |
File diff suppressed because it is too large
Load Diff
File diff suppressed because one or more lines are too long
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because one or more lines are too long
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
+22937
File diff suppressed because one or more lines are too long
+3822
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
+7271
File diff suppressed because one or more lines are too long
+1211
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,210 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Convert whisper.cpp JSON output to the compact Meeting Lab transcript shape."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
class WhisperCppConversionError(ValueError):
|
||||
"""Raised when whisper.cpp JSON cannot be converted deterministically."""
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Convert whisper.cpp JSON output to compact JSON and text files."
|
||||
)
|
||||
parser.add_argument("input_json", type=Path, help="whisper.cpp JSON file")
|
||||
parser.add_argument(
|
||||
"--json-output",
|
||||
type=Path,
|
||||
help="Compact JSON output path; default: <input>_converted.json",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--text-output",
|
||||
type=Path,
|
||||
help="Readable text output path; default: <input>_converted.txt",
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def load_json(path: Path) -> dict[str, Any]:
|
||||
data = json.loads(path.read_text(encoding="utf-8"))
|
||||
if not isinstance(data, dict):
|
||||
raise WhisperCppConversionError("Input JSON must contain a top-level object.")
|
||||
return data
|
||||
|
||||
|
||||
def transcription_entries(data: dict[str, Any]) -> list[Any]:
|
||||
entries = data.get("transcription")
|
||||
if not isinstance(entries, list):
|
||||
raise WhisperCppConversionError(
|
||||
"Input JSON must contain a 'transcription' list."
|
||||
)
|
||||
return entries
|
||||
|
||||
|
||||
def offset_seconds(entry: dict[str, Any], entry_index: int, field: str) -> float:
|
||||
offsets = entry.get("offsets")
|
||||
if not isinstance(offsets, dict):
|
||||
raise WhisperCppConversionError(
|
||||
f"transcription[{entry_index}].offsets must be an object."
|
||||
)
|
||||
value = offsets.get(field)
|
||||
if not isinstance(value, (int, float)):
|
||||
raise WhisperCppConversionError(
|
||||
f"transcription[{entry_index}].offsets.{field} must be a number."
|
||||
)
|
||||
return float(value) / 1000.0
|
||||
|
||||
|
||||
def validate_timestamp_fields(entry: dict[str, Any], entry_index: int) -> None:
|
||||
timestamps = entry.get("timestamps")
|
||||
if not isinstance(timestamps, dict):
|
||||
raise WhisperCppConversionError(
|
||||
f"transcription[{entry_index}].timestamps must be an object."
|
||||
)
|
||||
for field in ("from", "to"):
|
||||
if field not in timestamps:
|
||||
raise WhisperCppConversionError(
|
||||
f"transcription[{entry_index}].timestamps.{field} is missing."
|
||||
)
|
||||
|
||||
|
||||
def convert_entries(entries: list[Any]) -> dict[str, Any]:
|
||||
segments: list[dict[str, Any]] = []
|
||||
|
||||
for entry_index, entry in enumerate(entries):
|
||||
if not isinstance(entry, dict):
|
||||
raise WhisperCppConversionError(
|
||||
f"transcription[{entry_index}] must be an object."
|
||||
)
|
||||
validate_timestamp_fields(entry, entry_index)
|
||||
start = offset_seconds(entry, entry_index, "from")
|
||||
end = offset_seconds(entry, entry_index, "to")
|
||||
if end < start:
|
||||
raise WhisperCppConversionError(
|
||||
f"transcription[{entry_index}].offsets.to must be greater than or equal to offsets.from."
|
||||
)
|
||||
|
||||
text_value = entry.get("text", "")
|
||||
if not isinstance(text_value, str):
|
||||
raise WhisperCppConversionError(
|
||||
f"transcription[{entry_index}].text must be a string."
|
||||
)
|
||||
text = text_value.strip()
|
||||
if not text:
|
||||
continue
|
||||
|
||||
segments.append(
|
||||
{
|
||||
"id": len(segments),
|
||||
"start": start,
|
||||
"end": end,
|
||||
"text": text,
|
||||
}
|
||||
)
|
||||
|
||||
return {
|
||||
"text": " ".join(segment["text"] for segment in segments),
|
||||
"segments": segments,
|
||||
}
|
||||
|
||||
|
||||
def format_timestamp(seconds: float) -> str:
|
||||
milliseconds_total = int(round(seconds * 1000))
|
||||
milliseconds = milliseconds_total % 1000
|
||||
seconds_total = milliseconds_total // 1000
|
||||
second = seconds_total % 60
|
||||
minutes_total = seconds_total // 60
|
||||
minute = minutes_total % 60
|
||||
hour = minutes_total // 60
|
||||
return f"{hour:02d}:{minute:02d}:{second:02d}.{milliseconds:03d}"
|
||||
|
||||
|
||||
def render_text(segments: list[dict[str, Any]]) -> str:
|
||||
lines = [
|
||||
(
|
||||
f"[{format_timestamp(segment['start'])} - "
|
||||
f"{format_timestamp(segment['end'])}] {segment['text']}"
|
||||
)
|
||||
for segment in segments
|
||||
]
|
||||
return "\n".join(lines) + ("\n" if lines else "")
|
||||
|
||||
|
||||
def default_json_output(input_path: Path) -> Path:
|
||||
return input_path.with_name(f"{input_path.stem}_converted.json")
|
||||
|
||||
|
||||
def default_text_output(input_path: Path) -> Path:
|
||||
return input_path.with_name(f"{input_path.stem}_converted.txt")
|
||||
|
||||
|
||||
def convert_whispercpp_json(
|
||||
input_path: Path,
|
||||
json_output_path: Path | None = None,
|
||||
text_output_path: Path | None = None,
|
||||
) -> dict[str, Any]:
|
||||
data = load_json(input_path)
|
||||
entries = transcription_entries(data)
|
||||
converted = convert_entries(entries)
|
||||
json_output = json_output_path or default_json_output(input_path)
|
||||
text_output = text_output_path or default_text_output(input_path)
|
||||
|
||||
json_output.parent.mkdir(parents=True, exist_ok=True)
|
||||
text_output.parent.mkdir(parents=True, exist_ok=True)
|
||||
json_output.write_text(
|
||||
json.dumps(converted, ensure_ascii=False, indent=2) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
text_output.write_text(render_text(converted["segments"]), encoding="utf-8")
|
||||
|
||||
total_duration = (
|
||||
max((segment["end"] for segment in converted["segments"]), default=0.0)
|
||||
)
|
||||
summary = {
|
||||
"input_path": input_path,
|
||||
"json_output_path": json_output,
|
||||
"text_output_path": text_output,
|
||||
"original_entry_count": len(entries),
|
||||
"retained_segment_count": len(converted["segments"]),
|
||||
"total_character_count": len(converted["text"]),
|
||||
"detected_total_duration": total_duration,
|
||||
}
|
||||
print_summary(summary)
|
||||
return summary
|
||||
|
||||
|
||||
def print_summary(summary: dict[str, Any]) -> None:
|
||||
print(f"Input path: {summary['input_path']}")
|
||||
print(f"JSON output path: {summary['json_output_path']}")
|
||||
print(f"Text output path: {summary['text_output_path']}")
|
||||
print(f"Original entry count: {summary['original_entry_count']}")
|
||||
print(f"Retained segment count: {summary['retained_segment_count']}")
|
||||
print(f"Total character count: {summary['total_character_count']}")
|
||||
print(
|
||||
"Detected total duration: "
|
||||
f"{summary['detected_total_duration']:.3f} seconds"
|
||||
)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
try:
|
||||
convert_whispercpp_json(
|
||||
input_path=args.input_json,
|
||||
json_output_path=args.json_output,
|
||||
text_output_path=args.text_output,
|
||||
)
|
||||
except (OSError, UnicodeError, json.JSONDecodeError, WhisperCppConversionError) as exc:
|
||||
print(f"Error: {exc}")
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,145 @@
|
||||
import json
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
from scripts.convert_whispercpp_json import (
|
||||
WhisperCppConversionError,
|
||||
convert_entries,
|
||||
convert_whispercpp_json,
|
||||
format_timestamp,
|
||||
)
|
||||
|
||||
|
||||
def whispercpp_entry(
|
||||
text: str,
|
||||
offset_from: int,
|
||||
offset_to: int,
|
||||
timestamp_from: str = "00:00:00,000",
|
||||
timestamp_to: str = "00:00:01,000",
|
||||
) -> dict:
|
||||
return {
|
||||
"timestamps": {"from": timestamp_from, "to": timestamp_to},
|
||||
"offsets": {"from": offset_from, "to": offset_to},
|
||||
"text": text,
|
||||
}
|
||||
|
||||
|
||||
class ConvertWhisperCppJsonTests(unittest.TestCase):
|
||||
def test_valid_whispercpp_json_writes_compact_json_and_text(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
input_path = root / "whispercpp.json"
|
||||
json_output = root / "converted.json"
|
||||
text_output = root / "converted.txt"
|
||||
input_path.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"transcription": [
|
||||
whispercpp_entry(" Hello ", 0, 12340),
|
||||
whispercpp_entry("world", 12340, 13000),
|
||||
]
|
||||
},
|
||||
ensure_ascii=False,
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
summary = convert_whispercpp_json(input_path, json_output, text_output)
|
||||
|
||||
converted = json.loads(json_output.read_text(encoding="utf-8"))
|
||||
self.assertEqual(converted["text"], "Hello world")
|
||||
self.assertEqual(
|
||||
converted["segments"],
|
||||
[
|
||||
{"id": 0, "start": 0.0, "end": 12.34, "text": "Hello"},
|
||||
{"id": 1, "start": 12.34, "end": 13.0, "text": "world"},
|
||||
],
|
||||
)
|
||||
self.assertEqual(
|
||||
text_output.read_text(encoding="utf-8"),
|
||||
"[00:00:00.000 - 00:00:12.340] Hello\n"
|
||||
"[00:00:12.340 - 00:00:13.000] world\n",
|
||||
)
|
||||
self.assertEqual(summary["original_entry_count"], 2)
|
||||
self.assertEqual(summary["retained_segment_count"], 2)
|
||||
self.assertEqual(summary["total_character_count"], len("Hello world"))
|
||||
self.assertEqual(summary["detected_total_duration"], 13.0)
|
||||
|
||||
def test_empty_segments_are_skipped_and_ids_are_renumbered(self) -> None:
|
||||
converted = convert_entries(
|
||||
[
|
||||
whispercpp_entry("first", 0, 1000),
|
||||
whispercpp_entry(" ", 1000, 2000),
|
||||
whispercpp_entry("", 2000, 3000),
|
||||
whispercpp_entry("second", 3000, 4000),
|
||||
]
|
||||
)
|
||||
|
||||
self.assertEqual(converted["text"], "first second")
|
||||
self.assertEqual(
|
||||
converted["segments"],
|
||||
[
|
||||
{"id": 0, "start": 0.0, "end": 1.0, "text": "first"},
|
||||
{"id": 1, "start": 3.0, "end": 4.0, "text": "second"},
|
||||
],
|
||||
)
|
||||
|
||||
def test_missing_transcription_list_is_rejected(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
input_path = Path(directory) / "missing.json"
|
||||
input_path.write_text(json.dumps({"segments": []}), encoding="utf-8")
|
||||
|
||||
with self.assertRaisesRegex(
|
||||
WhisperCppConversionError,
|
||||
"Input JSON must contain a 'transcription' list.",
|
||||
):
|
||||
convert_whispercpp_json(input_path)
|
||||
|
||||
def test_malformed_offsets_are_rejected(self) -> None:
|
||||
with self.assertRaisesRegex(
|
||||
WhisperCppConversionError,
|
||||
r"transcription\[0\]\.offsets\.from must be a number.",
|
||||
):
|
||||
convert_entries([whispercpp_entry("bad", "0", 1000)])
|
||||
|
||||
with self.assertRaisesRegex(
|
||||
WhisperCppConversionError,
|
||||
r"offsets\.to must be greater than or equal to offsets\.from",
|
||||
):
|
||||
convert_entries([whispercpp_entry("bad", 2000, 1000)])
|
||||
|
||||
def test_utf8_text_is_preserved(self) -> None:
|
||||
converted = convert_entries(
|
||||
[
|
||||
whispercpp_entry(" Grüße für Marleen – HÜSKER ", 0, 1000),
|
||||
]
|
||||
)
|
||||
|
||||
self.assertEqual(converted["text"], "Grüße für Marleen – HÜSKER")
|
||||
self.assertEqual(
|
||||
converted["segments"][0]["text"],
|
||||
"Grüße für Marleen – HÜSKER",
|
||||
)
|
||||
|
||||
def test_ordering_is_preserved(self) -> None:
|
||||
converted = convert_entries(
|
||||
[
|
||||
whispercpp_entry("third spoken first", 3000, 4000),
|
||||
whispercpp_entry("first spoken second", 0, 1000),
|
||||
whispercpp_entry("second spoken third", 1000, 2000),
|
||||
]
|
||||
)
|
||||
|
||||
self.assertEqual(
|
||||
[segment["text"] for segment in converted["segments"]],
|
||||
["third spoken first", "first spoken second", "second spoken third"],
|
||||
)
|
||||
|
||||
def test_format_timestamp_uses_hours_minutes_seconds_milliseconds(self) -> None:
|
||||
self.assertEqual(format_timestamp(123.45), "00:02:03.450")
|
||||
self.assertEqual(format_timestamp(3723.004), "01:02:03.004")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user