From f00249db9d1ea051d29aa1bcca869fc4b88e83eb Mon Sep 17 00:00:00 2001 From: historia Date: Mon, 24 Aug 2026 02:59:26 -0400 Subject: refactor: add app directory, dir structure change --- app/tests/test_chunking.py | 118 +++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 118 insertions(+) create mode 100644 app/tests/test_chunking.py (limited to 'app/tests/test_chunking.py') diff --git a/app/tests/test_chunking.py b/app/tests/test_chunking.py new file mode 100644 index 0000000..2904e40 --- /dev/null +++ b/app/tests/test_chunking.py @@ -0,0 +1,118 @@ +"""Tests for text chunking.""" + +import unittest +from unittest.mock import patch + +from converter import config +from converter.chunking import split_into_chunks + + +class ChunkSizeDefaultTests(unittest.TestCase): + """Guard the request-size setting: each API call is one model + generation, and the servers silently truncate audio when a single + generation runs too long (~2.5 min faster backend, ~11 min Qwen + demo), so the default chunk size must stay well inside that budget. + There is no hard ceiling beyond CHUNK_SIZE; users raising it accept + the truncation risk themselves.""" + + def test_default_chunk_size_within_single_generation_budget(self): + self.assertLessEqual(config.CHUNK_SIZE, 300) + + def test_default_chunk_size_is_positive(self): + self.assertGreaterEqual(config.CHUNK_SIZE, 1) + + +class RequestSizeTests(unittest.TestCase): + def test_oversized_chunk_size_is_honored(self): + # No clamping: whatever size is configured (or requested) is used. + text = " ".join(f"word{i}" for i in range(30)) + "." + chunks = split_into_chunks(text, max_words=5000) + self.assertEqual(len(chunks), 1) + self.assertEqual(len(chunks[0].split()), 30) + + def test_default_uses_runtime_config_chunk_size(self): + # The default resolves config.CHUNK_SIZE at call time, so + # patching the config changes the default split size. + sentences = " ".join( + f"S{i} " + " ".join(["word"] * 8) + "." for i in range(60)) + with patch.object(config, "CHUNK_SIZE", 120): + chunks = split_into_chunks(sentences) + self.assertGreater(len(chunks), 1) + self.assertTrue(all(len(chunk.split()) <= 120 for chunk in chunks)) + + +class SplitIntoChunksTests(unittest.TestCase): + def test_empty_input(self): + self.assertEqual(split_into_chunks(""), []) + self.assertEqual(split_into_chunks(" \n "), []) + + def test_short_text_single_chunk(self): + self.assertEqual(split_into_chunks("One short sentence."), ["One short sentence."]) + + def test_respects_word_limit_across_sentences(self): + # 10 sentences of 9 words each = 90 words total + sentences = [f"S{i} " + " ".join(["word"] * 8) + "." for i in range(10)] + chunks = split_into_chunks(" ".join(sentences), max_words=25) + self.assertGreater(len(chunks), 1) + self.assertTrue(all(len(c.split()) <= 25 for c in chunks)) + self.assertEqual(sum(len(c.split()) for c in chunks), 90) + + def test_long_sentence_split_keeps_punctuation(self): + # 10 clauses of 5 words each, joined by comma+space + sentence = ", ".join([" ".join(["w"] * 5) for _ in range(10)]) + "." + chunks = split_into_chunks(sentence, max_words=12) + self.assertGreater(len(chunks), 1) + self.assertTrue(all(len(c.split()) <= 12 for c in chunks)) + self.assertIn(",", chunks[0]) # commas retained for TTS prosody + + def test_clause_split_never_breaks_numbers(self): + # Regression: the clause split used to fire at every comma even + # without whitespace, mutating "1,000,000" into "1, 000, 000". + sentence = ("There were exactly 1,000,000 soldiers marching at 12:30, " + + "and they kept marching onward " * 30) + "endlessly." + chunks = split_into_chunks(sentence, max_words=25) + self.assertGreater(len(chunks), 1) + joined = " ".join(chunks) + self.assertIn("1,000,000", joined) + self.assertIn("12:30", joined) + self.assertNotIn("1, 000", joined) + self.assertNotIn("000, 000", joined) + self.assertNotIn("12: 30", joined) + + def test_clause_split_requires_whitespace_after_punctuation(self): + # Run-on clauses without spaces after commas have no clause split + # point, so the last-resort word-boundary split fires instead. + # Tokens themselves (and numbers like "1,000,000") stay intact. + sentence = ",".join([" ".join(["w"] * 5) for _ in range(10)]) + "." + chunks = split_into_chunks(sentence, max_words=12) + self.assertGreater(len(chunks), 1) + self.assertTrue(all(len(chunk.split()) <= 12 for chunk in chunks)) + tokens = sentence.replace(",", " , ").split() + rejoined = " ".join(chunks).replace(",", " , ").split() + self.assertEqual(rejoined, tokens) + + def test_single_oversized_sentence_is_word_split(self): + # A punctuation-free sentence longer than the limit is split at word + # boundaries so no single request exceeds the configured size. + sentence = " ".join(["word"] * 30) + "." + chunks = split_into_chunks(sentence, max_words=10) + self.assertGreater(len(chunks), 1) + self.assertTrue(all(len(chunk.split()) <= 10 for chunk in chunks)) + self.assertEqual(sum(len(chunk.split()) for chunk in chunks), 30) + + def test_word_split_never_breaks_number_tokens(self): + # Numbers and other punctuation-bearing tokens are single words and + # must never be broken apart by the last-resort word split. + sentence = ("There were exactly 1,000,000 soldiers marching at 12:30 " + "and " + "they kept marching onward " * 20) + "endlessly." + chunks = split_into_chunks(sentence, max_words=10) + self.assertGreater(len(chunks), 1) + joined = " ".join(chunks) + self.assertIn("1,000,000", joined) + self.assertIn("12:30", joined) + self.assertNotIn("1, 000", joined) + self.assertNotIn("12: 30", joined) + + +if __name__ == "__main__": + unittest.main() -- cgit v1.2.3