aboutsummaryrefslogtreecommitdiff
path: root/tests/test_chunking.py
blob: da45f650a15add3949a4c642406ecbd5d9c8213c (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
"""Tests for text chunking."""

import unittest

from converter.chunking import split_into_chunks


class SplitIntoChunksTests(unittest.TestCase):
    def test_empty_input(self):
        self.assertEqual(split_into_chunks(""), [])
        self.assertEqual(split_into_chunks("   \n  "), [])

    def test_short_text_single_chunk(self):
        self.assertEqual(split_into_chunks("One short sentence."), ["One short sentence."])

    def test_respects_word_limit_across_sentences(self):
        # 10 sentences of 9 words each = 90 words total
        sentences = [f"S{i} " + " ".join(["word"] * 8) + "." for i in range(10)]
        chunks = split_into_chunks(" ".join(sentences), max_words=25)
        self.assertGreater(len(chunks), 1)
        self.assertTrue(all(len(c.split()) <= 25 for c in chunks))
        self.assertEqual(sum(len(c.split()) for c in chunks), 90)

    def test_long_sentence_split_keeps_punctuation(self):
        # 10 clauses of 5 words each, joined by commas
        sentence = ",".join([" ".join(["w"] * 5) for _ in range(10)]) + "."
        chunks = split_into_chunks(sentence, max_words=12)
        self.assertGreater(len(chunks), 1)
        self.assertTrue(all(len(c.split()) <= 12 for c in chunks))
        self.assertIn(",", chunks[0])  # commas retained for TTS prosody

    def test_single_oversized_sentence_stays_intact(self):
        sentence = " ".join(["word"] * 30) + "."
        chunks = split_into_chunks(sentence, max_words=10)
        self.assertEqual(chunks, [sentence])


if __name__ == "__main__":
    unittest.main()