aboutsummaryrefslogtreecommitdiff
path: root/app/tests/test_chunking.py
diff options
context:
space:
mode:
authorhistoria <historiavg@proton.me>2026-08-24 02:59:26 -0400
committerhistoria <historiavg@proton.me>2026-08-24 02:59:26 -0400
commitf00249db9d1ea051d29aa1bcca869fc4b88e83eb (patch)
treea75f076fac1b63e0b4bf2eb8f54affbcc681a891 /app/tests/test_chunking.py
parent9dd4f9595be3b1d76a3a07dc3eca90cfaf8a3f97 (diff)
downloadtts-audiobook-generator-f00249db9d1ea051d29aa1bcca869fc4b88e83eb.tar.gz
refactor: add app directory, dir structure change
Diffstat (limited to 'app/tests/test_chunking.py')
-rw-r--r--app/tests/test_chunking.py118
1 files changed, 118 insertions, 0 deletions
diff --git a/app/tests/test_chunking.py b/app/tests/test_chunking.py
new file mode 100644
index 0000000..2904e40
--- /dev/null
+++ b/app/tests/test_chunking.py
@@ -0,0 +1,118 @@
+"""Tests for text chunking."""
+
+import unittest
+from unittest.mock import patch
+
+from converter import config
+from converter.chunking import split_into_chunks
+
+
+class ChunkSizeDefaultTests(unittest.TestCase):
+ """Guard the request-size setting: each API call is one model
+ generation, and the servers silently truncate audio when a single
+ generation runs too long (~2.5 min faster backend, ~11 min Qwen
+ demo), so the default chunk size must stay well inside that budget.
+ There is no hard ceiling beyond CHUNK_SIZE; users raising it accept
+ the truncation risk themselves."""
+
+ def test_default_chunk_size_within_single_generation_budget(self):
+ self.assertLessEqual(config.CHUNK_SIZE, 300)
+
+ def test_default_chunk_size_is_positive(self):
+ self.assertGreaterEqual(config.CHUNK_SIZE, 1)
+
+
+class RequestSizeTests(unittest.TestCase):
+ def test_oversized_chunk_size_is_honored(self):
+ # No clamping: whatever size is configured (or requested) is used.
+ text = " ".join(f"word{i}" for i in range(30)) + "."
+ chunks = split_into_chunks(text, max_words=5000)
+ self.assertEqual(len(chunks), 1)
+ self.assertEqual(len(chunks[0].split()), 30)
+
+ def test_default_uses_runtime_config_chunk_size(self):
+ # The default resolves config.CHUNK_SIZE at call time, so
+ # patching the config changes the default split size.
+ sentences = " ".join(
+ f"S{i} " + " ".join(["word"] * 8) + "." for i in range(60))
+ with patch.object(config, "CHUNK_SIZE", 120):
+ chunks = split_into_chunks(sentences)
+ self.assertGreater(len(chunks), 1)
+ self.assertTrue(all(len(chunk.split()) <= 120 for chunk in chunks))
+
+
+class SplitIntoChunksTests(unittest.TestCase):
+ def test_empty_input(self):
+ self.assertEqual(split_into_chunks(""), [])
+ self.assertEqual(split_into_chunks(" \n "), [])
+
+ def test_short_text_single_chunk(self):
+ self.assertEqual(split_into_chunks("One short sentence."), ["One short sentence."])
+
+ def test_respects_word_limit_across_sentences(self):
+ # 10 sentences of 9 words each = 90 words total
+ sentences = [f"S{i} " + " ".join(["word"] * 8) + "." for i in range(10)]
+ chunks = split_into_chunks(" ".join(sentences), max_words=25)
+ self.assertGreater(len(chunks), 1)
+ self.assertTrue(all(len(c.split()) <= 25 for c in chunks))
+ self.assertEqual(sum(len(c.split()) for c in chunks), 90)
+
+ def test_long_sentence_split_keeps_punctuation(self):
+ # 10 clauses of 5 words each, joined by comma+space
+ sentence = ", ".join([" ".join(["w"] * 5) for _ in range(10)]) + "."
+ chunks = split_into_chunks(sentence, max_words=12)
+ self.assertGreater(len(chunks), 1)
+ self.assertTrue(all(len(c.split()) <= 12 for c in chunks))
+ self.assertIn(",", chunks[0]) # commas retained for TTS prosody
+
+ def test_clause_split_never_breaks_numbers(self):
+ # Regression: the clause split used to fire at every comma even
+ # without whitespace, mutating "1,000,000" into "1, 000, 000".
+ sentence = ("There were exactly 1,000,000 soldiers marching at 12:30, "
+ + "and they kept marching onward " * 30) + "endlessly."
+ chunks = split_into_chunks(sentence, max_words=25)
+ self.assertGreater(len(chunks), 1)
+ joined = " ".join(chunks)
+ self.assertIn("1,000,000", joined)
+ self.assertIn("12:30", joined)
+ self.assertNotIn("1, 000", joined)
+ self.assertNotIn("000, 000", joined)
+ self.assertNotIn("12: 30", joined)
+
+ def test_clause_split_requires_whitespace_after_punctuation(self):
+ # Run-on clauses without spaces after commas have no clause split
+ # point, so the last-resort word-boundary split fires instead.
+ # Tokens themselves (and numbers like "1,000,000") stay intact.
+ sentence = ",".join([" ".join(["w"] * 5) for _ in range(10)]) + "."
+ chunks = split_into_chunks(sentence, max_words=12)
+ self.assertGreater(len(chunks), 1)
+ self.assertTrue(all(len(chunk.split()) <= 12 for chunk in chunks))
+ tokens = sentence.replace(",", " , ").split()
+ rejoined = " ".join(chunks).replace(",", " , ").split()
+ self.assertEqual(rejoined, tokens)
+
+ def test_single_oversized_sentence_is_word_split(self):
+ # A punctuation-free sentence longer than the limit is split at word
+ # boundaries so no single request exceeds the configured size.
+ sentence = " ".join(["word"] * 30) + "."
+ chunks = split_into_chunks(sentence, max_words=10)
+ self.assertGreater(len(chunks), 1)
+ self.assertTrue(all(len(chunk.split()) <= 10 for chunk in chunks))
+ self.assertEqual(sum(len(chunk.split()) for chunk in chunks), 30)
+
+ def test_word_split_never_breaks_number_tokens(self):
+ # Numbers and other punctuation-bearing tokens are single words and
+ # must never be broken apart by the last-resort word split.
+ sentence = ("There were exactly 1,000,000 soldiers marching at 12:30 "
+ "and " + "they kept marching onward " * 20) + "endlessly."
+ chunks = split_into_chunks(sentence, max_words=10)
+ self.assertGreater(len(chunks), 1)
+ joined = " ".join(chunks)
+ self.assertIn("1,000,000", joined)
+ self.assertIn("12:30", joined)
+ self.assertNotIn("1, 000", joined)
+ self.assertNotIn("12: 30", joined)
+
+
+if __name__ == "__main__":
+ unittest.main()