51 lines
1.5 KiB
Python
51 lines
1.5 KiB
Python
import json
|
|
import tempfile
|
|
import unittest
|
|
from pathlib import Path
|
|
|
|
from src.meeting_lab.chunking.chunk_transcript import (
|
|
build_chunks,
|
|
read_transcript_blocks,
|
|
rendered_length,
|
|
)
|
|
|
|
|
|
class ChunkTranscriptTests(unittest.TestCase):
|
|
def test_whisper_json_uses_segments_not_aggregate_text(self) -> None:
|
|
data = {
|
|
"text": "alpha beta gamma",
|
|
"segments": [
|
|
{"text": "alpha"},
|
|
{"text": "beta"},
|
|
{"text": "gamma"},
|
|
],
|
|
}
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
input_file = Path(directory) / "meeting.json"
|
|
input_file.write_text(json.dumps(data), encoding="utf-8")
|
|
|
|
blocks = read_transcript_blocks(input_file)
|
|
|
|
self.assertEqual(blocks, ["alpha", "beta", "gamma"])
|
|
self.assertNotIn("alpha beta gamma", blocks)
|
|
|
|
def test_chunks_do_not_share_later_blocks_without_overlap(self) -> None:
|
|
blocks = [f"block-{number:02d}-" + ("x" * 20) for number in range(10)]
|
|
|
|
chunks = build_chunks(
|
|
blocks=blocks,
|
|
target_chars=60,
|
|
max_chars=80,
|
|
min_chars=30,
|
|
overlap_blocks=0,
|
|
)
|
|
|
|
flattened = [block for chunk in chunks for block in chunk]
|
|
self.assertEqual(flattened, blocks)
|
|
self.assertEqual(len(flattened), len(set(flattened)))
|
|
self.assertTrue(all(rendered_length(chunk) <= 80 for chunk in chunks))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|