"""Tests for text chunking.""" import unittest from unittest.mock import patch from converter import config from converter.chunking import split_into_chunks class ChunkSizeDefaultTests(unittest.TestCase): """Guard the request-size setting: each API call is one model generation, and the servers silently truncate audio when a single generation runs too long (~2.5 min faster backend, ~11 min Gradio demo), so the default chunk size must stay well inside that budget. There is no hard ceiling beyond CHUNK_SIZE; users raising it accept the truncation risk themselves.""" def test_default_chunk_size_within_single_generation_budget(self): self.assertLessEqual(config.CHUNK_SIZE, 300) def test_default_chunk_size_is_positive(self): self.assertGreaterEqual(config.CHUNK_SIZE, 1) class RequestSizeTests(unittest.TestCase): def test_oversized_chunk_size_is_honored(self): # No clamping: whatever size is configured (or requested) is used. text = " ".join(f"word{i}" for i in range(30)) + "." chunks = split_into_chunks(text, max_words=5000) self.assertEqual(len(chunks), 1) self.assertEqual(len(chunks[0].split()), 30) def test_default_uses_runtime_config_chunk_size(self): # The default resolves config.CHUNK_SIZE at call time, so # patching the config changes the default split size. sentences = " ".join( f"S{i} " + " ".join(["word"] * 8) + "." for i in range(60)) with patch.object(config, "CHUNK_SIZE", 120): chunks = split_into_chunks(sentences) self.assertGreater(len(chunks), 1) self.assertTrue(all(len(chunk.split()) <= 120 for chunk in chunks)) class SplitIntoChunksTests(unittest.TestCase): def test_empty_input(self): self.assertEqual(split_into_chunks(""), []) self.assertEqual(split_into_chunks(" \n "), []) def test_short_text_single_chunk(self): self.assertEqual(split_into_chunks("One short sentence."), ["One short sentence."]) def test_respects_word_limit_across_sentences(self): # 10 sentences of 9 words each = 90 words total sentences = [f"S{i} " + " ".join(["word"] * 8) + "." for i in range(10)] chunks = split_into_chunks(" ".join(sentences), max_words=25) self.assertGreater(len(chunks), 1) self.assertTrue(all(len(c.split()) <= 25 for c in chunks)) self.assertEqual(sum(len(c.split()) for c in chunks), 90) def test_long_sentence_split_keeps_punctuation(self): # 10 clauses of 5 words each, joined by comma+space sentence = ", ".join([" ".join(["w"] * 5) for _ in range(10)]) + "." chunks = split_into_chunks(sentence, max_words=12) self.assertGreater(len(chunks), 1) self.assertTrue(all(len(c.split()) <= 12 for c in chunks)) self.assertIn(",", chunks[0]) # commas retained for TTS prosody def test_clause_split_never_breaks_numbers(self): # Regression: the clause split used to fire at every comma even # without whitespace, mutating "1,000,000" into "1, 000, 000". sentence = ("There were exactly 1,000,000 soldiers marching at 12:30, " + "and they kept marching onward " * 30) + "endlessly." chunks = split_into_chunks(sentence, max_words=25) self.assertGreater(len(chunks), 1) joined = " ".join(chunks) self.assertIn("1,000,000", joined) self.assertIn("12:30", joined) self.assertNotIn("1, 000", joined) self.assertNotIn("000, 000", joined) self.assertNotIn("12: 30", joined) def test_clause_split_requires_whitespace_after_punctuation(self): # Run-on clauses without spaces after commas have no clause split # point, so the last-resort word-boundary split fires instead. # Tokens themselves (and numbers like "1,000,000") stay intact. sentence = ",".join([" ".join(["w"] * 5) for _ in range(10)]) + "." chunks = split_into_chunks(sentence, max_words=12) self.assertGreater(len(chunks), 1) self.assertTrue(all(len(chunk.split()) <= 12 for chunk in chunks)) tokens = sentence.replace(",", " , ").split() rejoined = " ".join(chunks).replace(",", " , ").split() self.assertEqual(rejoined, tokens) def test_single_oversized_sentence_is_word_split(self): # A punctuation-free sentence longer than the limit is split at word # boundaries so no single request exceeds the configured size. sentence = " ".join(["word"] * 30) + "." chunks = split_into_chunks(sentence, max_words=10) self.assertGreater(len(chunks), 1) self.assertTrue(all(len(chunk.split()) <= 10 for chunk in chunks)) self.assertEqual(sum(len(chunk.split()) for chunk in chunks), 30) def test_word_split_never_breaks_number_tokens(self): # Numbers and other punctuation-bearing tokens are single words and # must never be broken apart by the last-resort word split. sentence = ("There were exactly 1,000,000 soldiers marching at 12:30 " "and " + "they kept marching onward " * 20) + "endlessly." chunks = split_into_chunks(sentence, max_words=10) self.assertGreater(len(chunks), 1) joined = " ".join(chunks) self.assertIn("1,000,000", joined) self.assertIn("12:30", joined) self.assertNotIn("1, 000", joined) self.assertNotIn("12: 30", joined) if __name__ == "__main__": unittest.main()