From f00249db9d1ea051d29aa1bcca869fc4b88e83eb Mon Sep 17 00:00:00 2001 From: historia Date: Mon, 24 Aug 2026 02:59:26 -0400 Subject: refactor: add app directory, dir structure change --- tests/test_chunking.py | 118 ------------------------------------------------- 1 file changed, 118 deletions(-) delete mode 100644 tests/test_chunking.py (limited to 'tests/test_chunking.py') diff --git a/tests/test_chunking.py b/tests/test_chunking.py deleted file mode 100644 index 2904e40..0000000 --- a/tests/test_chunking.py +++ /dev/null @@ -1,118 +0,0 @@ -"""Tests for text chunking.""" - -import unittest -from unittest.mock import patch - -from converter import config -from converter.chunking import split_into_chunks - - -class ChunkSizeDefaultTests(unittest.TestCase): - """Guard the request-size setting: each API call is one model - generation, and the servers silently truncate audio when a single - generation runs too long (~2.5 min faster backend, ~11 min Qwen - demo), so the default chunk size must stay well inside that budget. - There is no hard ceiling beyond CHUNK_SIZE; users raising it accept - the truncation risk themselves.""" - - def test_default_chunk_size_within_single_generation_budget(self): - self.assertLessEqual(config.CHUNK_SIZE, 300) - - def test_default_chunk_size_is_positive(self): - self.assertGreaterEqual(config.CHUNK_SIZE, 1) - - -class RequestSizeTests(unittest.TestCase): - def test_oversized_chunk_size_is_honored(self): - # No clamping: whatever size is configured (or requested) is used. - text = " ".join(f"word{i}" for i in range(30)) + "." - chunks = split_into_chunks(text, max_words=5000) - self.assertEqual(len(chunks), 1) - self.assertEqual(len(chunks[0].split()), 30) - - def test_default_uses_runtime_config_chunk_size(self): - # The default resolves config.CHUNK_SIZE at call time, so - # patching the config changes the default split size. - sentences = " ".join( - f"S{i} " + " ".join(["word"] * 8) + "." for i in range(60)) - with patch.object(config, "CHUNK_SIZE", 120): - chunks = split_into_chunks(sentences) - self.assertGreater(len(chunks), 1) - self.assertTrue(all(len(chunk.split()) <= 120 for chunk in chunks)) - - -class SplitIntoChunksTests(unittest.TestCase): - def test_empty_input(self): - self.assertEqual(split_into_chunks(""), []) - self.assertEqual(split_into_chunks(" \n "), []) - - def test_short_text_single_chunk(self): - self.assertEqual(split_into_chunks("One short sentence."), ["One short sentence."]) - - def test_respects_word_limit_across_sentences(self): - # 10 sentences of 9 words each = 90 words total - sentences = [f"S{i} " + " ".join(["word"] * 8) + "." for i in range(10)] - chunks = split_into_chunks(" ".join(sentences), max_words=25) - self.assertGreater(len(chunks), 1) - self.assertTrue(all(len(c.split()) <= 25 for c in chunks)) - self.assertEqual(sum(len(c.split()) for c in chunks), 90) - - def test_long_sentence_split_keeps_punctuation(self): - # 10 clauses of 5 words each, joined by comma+space - sentence = ", ".join([" ".join(["w"] * 5) for _ in range(10)]) + "." - chunks = split_into_chunks(sentence, max_words=12) - self.assertGreater(len(chunks), 1) - self.assertTrue(all(len(c.split()) <= 12 for c in chunks)) - self.assertIn(",", chunks[0]) # commas retained for TTS prosody - - def test_clause_split_never_breaks_numbers(self): - # Regression: the clause split used to fire at every comma even - # without whitespace, mutating "1,000,000" into "1, 000, 000". - sentence = ("There were exactly 1,000,000 soldiers marching at 12:30, " - + "and they kept marching onward " * 30) + "endlessly." - chunks = split_into_chunks(sentence, max_words=25) - self.assertGreater(len(chunks), 1) - joined = " ".join(chunks) - self.assertIn("1,000,000", joined) - self.assertIn("12:30", joined) - self.assertNotIn("1, 000", joined) - self.assertNotIn("000, 000", joined) - self.assertNotIn("12: 30", joined) - - def test_clause_split_requires_whitespace_after_punctuation(self): - # Run-on clauses without spaces after commas have no clause split - # point, so the last-resort word-boundary split fires instead. - # Tokens themselves (and numbers like "1,000,000") stay intact. - sentence = ",".join([" ".join(["w"] * 5) for _ in range(10)]) + "." - chunks = split_into_chunks(sentence, max_words=12) - self.assertGreater(len(chunks), 1) - self.assertTrue(all(len(chunk.split()) <= 12 for chunk in chunks)) - tokens = sentence.replace(",", " , ").split() - rejoined = " ".join(chunks).replace(",", " , ").split() - self.assertEqual(rejoined, tokens) - - def test_single_oversized_sentence_is_word_split(self): - # A punctuation-free sentence longer than the limit is split at word - # boundaries so no single request exceeds the configured size. - sentence = " ".join(["word"] * 30) + "." - chunks = split_into_chunks(sentence, max_words=10) - self.assertGreater(len(chunks), 1) - self.assertTrue(all(len(chunk.split()) <= 10 for chunk in chunks)) - self.assertEqual(sum(len(chunk.split()) for chunk in chunks), 30) - - def test_word_split_never_breaks_number_tokens(self): - # Numbers and other punctuation-bearing tokens are single words and - # must never be broken apart by the last-resort word split. - sentence = ("There were exactly 1,000,000 soldiers marching at 12:30 " - "and " + "they kept marching onward " * 20) + "endlessly." - chunks = split_into_chunks(sentence, max_words=10) - self.assertGreater(len(chunks), 1) - joined = " ".join(chunks) - self.assertIn("1,000,000", joined) - self.assertIn("12:30", joined) - self.assertNotIn("1, 000", joined) - self.assertNotIn("12: 30", joined) - - -if __name__ == "__main__": - unittest.main() -- cgit v1.2.3