"""Tests for text chunking.""" import unittest from converter.chunking import split_into_chunks class SplitIntoChunksTests(unittest.TestCase): def test_empty_input(self): self.assertEqual(split_into_chunks(""), []) self.assertEqual(split_into_chunks(" \n "), []) def test_short_text_single_chunk(self): self.assertEqual(split_into_chunks("One short sentence."), ["One short sentence."]) def test_respects_word_limit_across_sentences(self): # 10 sentences of 9 words each = 90 words total sentences = [f"S{i} " + " ".join(["word"] * 8) + "." for i in range(10)] chunks = split_into_chunks(" ".join(sentences), max_words=25) self.assertGreater(len(chunks), 1) self.assertTrue(all(len(c.split()) <= 25 for c in chunks)) self.assertEqual(sum(len(c.split()) for c in chunks), 90) def test_long_sentence_split_keeps_punctuation(self): # 10 clauses of 5 words each, joined by commas sentence = ",".join([" ".join(["w"] * 5) for _ in range(10)]) + "." chunks = split_into_chunks(sentence, max_words=12) self.assertGreater(len(chunks), 1) self.assertTrue(all(len(c.split()) <= 12 for c in chunks)) self.assertIn(",", chunks[0]) # commas retained for TTS prosody def test_single_oversized_sentence_stays_intact(self): sentence = " ".join(["word"] * 30) + "." chunks = split_into_chunks(sentence, max_words=10) self.assertEqual(chunks, [sentence]) if __name__ == "__main__": unittest.main()