aboutsummaryrefslogtreecommitdiff
path: root/tests/test_chunking.py
diff options
context:
space:
mode:
Diffstat (limited to 'tests/test_chunking.py')
-rw-r--r--tests/test_chunking.py118
1 files changed, 0 insertions, 118 deletions
diff --git a/tests/test_chunking.py b/tests/test_chunking.py
deleted file mode 100644
index 2904e40..0000000
--- a/tests/test_chunking.py
+++ /dev/null
@@ -1,118 +0,0 @@
-"""Tests for text chunking."""
-
-import unittest
-from unittest.mock import patch
-
-from converter import config
-from converter.chunking import split_into_chunks
-
-
-class ChunkSizeDefaultTests(unittest.TestCase):
- """Guard the request-size setting: each API call is one model
- generation, and the servers silently truncate audio when a single
- generation runs too long (~2.5 min faster backend, ~11 min Qwen
- demo), so the default chunk size must stay well inside that budget.
- There is no hard ceiling beyond CHUNK_SIZE; users raising it accept
- the truncation risk themselves."""
-
- def test_default_chunk_size_within_single_generation_budget(self):
- self.assertLessEqual(config.CHUNK_SIZE, 300)
-
- def test_default_chunk_size_is_positive(self):
- self.assertGreaterEqual(config.CHUNK_SIZE, 1)
-
-
-class RequestSizeTests(unittest.TestCase):
- def test_oversized_chunk_size_is_honored(self):
- # No clamping: whatever size is configured (or requested) is used.
- text = " ".join(f"word{i}" for i in range(30)) + "."
- chunks = split_into_chunks(text, max_words=5000)
- self.assertEqual(len(chunks), 1)
- self.assertEqual(len(chunks[0].split()), 30)
-
- def test_default_uses_runtime_config_chunk_size(self):
- # The default resolves config.CHUNK_SIZE at call time, so
- # patching the config changes the default split size.
- sentences = " ".join(
- f"S{i} " + " ".join(["word"] * 8) + "." for i in range(60))
- with patch.object(config, "CHUNK_SIZE", 120):
- chunks = split_into_chunks(sentences)
- self.assertGreater(len(chunks), 1)
- self.assertTrue(all(len(chunk.split()) <= 120 for chunk in chunks))
-
-
-class SplitIntoChunksTests(unittest.TestCase):
- def test_empty_input(self):
- self.assertEqual(split_into_chunks(""), [])
- self.assertEqual(split_into_chunks(" \n "), [])
-
- def test_short_text_single_chunk(self):
- self.assertEqual(split_into_chunks("One short sentence."), ["One short sentence."])
-
- def test_respects_word_limit_across_sentences(self):
- # 10 sentences of 9 words each = 90 words total
- sentences = [f"S{i} " + " ".join(["word"] * 8) + "." for i in range(10)]
- chunks = split_into_chunks(" ".join(sentences), max_words=25)
- self.assertGreater(len(chunks), 1)
- self.assertTrue(all(len(c.split()) <= 25 for c in chunks))
- self.assertEqual(sum(len(c.split()) for c in chunks), 90)
-
- def test_long_sentence_split_keeps_punctuation(self):
- # 10 clauses of 5 words each, joined by comma+space
- sentence = ", ".join([" ".join(["w"] * 5) for _ in range(10)]) + "."
- chunks = split_into_chunks(sentence, max_words=12)
- self.assertGreater(len(chunks), 1)
- self.assertTrue(all(len(c.split()) <= 12 for c in chunks))
- self.assertIn(",", chunks[0]) # commas retained for TTS prosody
-
- def test_clause_split_never_breaks_numbers(self):
- # Regression: the clause split used to fire at every comma even
- # without whitespace, mutating "1,000,000" into "1, 000, 000".
- sentence = ("There were exactly 1,000,000 soldiers marching at 12:30, "
- + "and they kept marching onward " * 30) + "endlessly."
- chunks = split_into_chunks(sentence, max_words=25)
- self.assertGreater(len(chunks), 1)
- joined = " ".join(chunks)
- self.assertIn("1,000,000", joined)
- self.assertIn("12:30", joined)
- self.assertNotIn("1, 000", joined)
- self.assertNotIn("000, 000", joined)
- self.assertNotIn("12: 30", joined)
-
- def test_clause_split_requires_whitespace_after_punctuation(self):
- # Run-on clauses without spaces after commas have no clause split
- # point, so the last-resort word-boundary split fires instead.
- # Tokens themselves (and numbers like "1,000,000") stay intact.
- sentence = ",".join([" ".join(["w"] * 5) for _ in range(10)]) + "."
- chunks = split_into_chunks(sentence, max_words=12)
- self.assertGreater(len(chunks), 1)
- self.assertTrue(all(len(chunk.split()) <= 12 for chunk in chunks))
- tokens = sentence.replace(",", " , ").split()
- rejoined = " ".join(chunks).replace(",", " , ").split()
- self.assertEqual(rejoined, tokens)
-
- def test_single_oversized_sentence_is_word_split(self):
- # A punctuation-free sentence longer than the limit is split at word
- # boundaries so no single request exceeds the configured size.
- sentence = " ".join(["word"] * 30) + "."
- chunks = split_into_chunks(sentence, max_words=10)
- self.assertGreater(len(chunks), 1)
- self.assertTrue(all(len(chunk.split()) <= 10 for chunk in chunks))
- self.assertEqual(sum(len(chunk.split()) for chunk in chunks), 30)
-
- def test_word_split_never_breaks_number_tokens(self):
- # Numbers and other punctuation-bearing tokens are single words and
- # must never be broken apart by the last-resort word split.
- sentence = ("There were exactly 1,000,000 soldiers marching at 12:30 "
- "and " + "they kept marching onward " * 20) + "endlessly."
- chunks = split_into_chunks(sentence, max_words=10)
- self.assertGreater(len(chunks), 1)
- joined = " ".join(chunks)
- self.assertIn("1,000,000", joined)
- self.assertIn("12:30", joined)
- self.assertNotIn("1, 000", joined)
- self.assertNotIn("12: 30", joined)
-
-
-if __name__ == "__main__":
- unittest.main()