aboutsummaryrefslogtreecommitdiff
path: root/tests/test_chunking.py
blob: 659a7715e622be0a30951998291d9d8decf2a1f5 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
"""Tests for text chunking."""

import unittest

from converter.chunking import split_into_chunks


class SplitIntoChunksTests(unittest.TestCase):
    def test_empty_input(self):
        self.assertEqual(split_into_chunks(""), [])
        self.assertEqual(split_into_chunks("   \n  "), [])

    def test_short_text_single_chunk(self):
        self.assertEqual(split_into_chunks("One short sentence."), ["One short sentence."])

    def test_respects_word_limit_across_sentences(self):
        # 10 sentences of 9 words each = 90 words total
        sentences = [f"S{i} " + " ".join(["word"] * 8) + "." for i in range(10)]
        chunks = split_into_chunks(" ".join(sentences), max_words=25)
        self.assertGreater(len(chunks), 1)
        self.assertTrue(all(len(c.split()) <= 25 for c in chunks))
        self.assertEqual(sum(len(c.split()) for c in chunks), 90)

    def test_long_sentence_split_keeps_punctuation(self):
        # 10 clauses of 5 words each, joined by comma+space
        sentence = ", ".join([" ".join(["w"] * 5) for _ in range(10)]) + "."
        chunks = split_into_chunks(sentence, max_words=12)
        self.assertGreater(len(chunks), 1)
        self.assertTrue(all(len(c.split()) <= 12 for c in chunks))
        self.assertIn(",", chunks[0])  # commas retained for TTS prosody

    def test_clause_split_never_breaks_numbers(self):
        # Regression: the clause split used to fire at every comma even
        # without whitespace, mutating "1,000,000" into "1, 000, 000".
        sentence = ("There were exactly 1,000,000 soldiers marching at 12:30, "
                    + "and they kept marching onward " * 30) + "endlessly."
        chunks = split_into_chunks(sentence, max_words=25)
        self.assertGreater(len(chunks), 1)
        joined = " ".join(chunks)
        self.assertIn("1,000,000", joined)
        self.assertIn("12:30", joined)
        self.assertNotIn("1, 000", joined)
        self.assertNotIn("000, 000", joined)
        self.assertNotIn("12: 30", joined)

    def test_clause_split_requires_whitespace_after_punctuation(self):
        # Run-on clauses without spaces after commas have no split point and
        # must stay byte-identical rather than being re-joined with spaces.
        sentence = ",".join([" ".join(["w"] * 5) for _ in range(10)]) + "."
        chunks = split_into_chunks(sentence, max_words=12)
        self.assertEqual(chunks, [sentence])

    def test_single_oversized_sentence_stays_intact(self):
        sentence = " ".join(["word"] * 30) + "."
        chunks = split_into_chunks(sentence, max_words=10)
        self.assertEqual(chunks, [sentence])


if __name__ == "__main__":
    unittest.main()