1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
|
"""Split extracted book text into TTS-sized chunks."""
import logging
import re
from typing import List, Optional
from . import config
logger = logging.getLogger(__name__)
def split_into_chunks(text: str, max_words: Optional[int] = None) -> List[str]:
"""Split text into chunks of at most ``max_words`` words.
``max_words`` defaults to ``config.CHUNK_SIZE`` (read at call time).
There is no ceiling beyond that setting, but note that the TTS
servers silently truncate audio when a single generation runs too
long without reporting an error, so very large values are at your
own risk (see CHUNK_SIZE in converter/config.py).
Splits on sentence boundaries. Sentences longer than the limit are
split further at clause punctuation (which is kept attached for TTS
prosody). Clause splits only happen at whitespace after punctuation,
so tokens like "1,000,000" or "12:30" are never broken apart. A piece
with no usable punctuation split point longer than the limit is split
at word boundaries as a last resort: individual tokens stay intact,
but whitespace between them is normalized.
"""
if max_words is None:
max_words = config.CHUNK_SIZE
if max_words < 1:
max_words = 1
if not text.strip():
return []
sentences = re.split(r"(?<=[.!?])\s+", text)
chunks = []
current_chunk = ""
current_words = 0
for sentence in sentences:
sentence_words = len(sentence.split())
if sentence_words > max_words:
if current_chunk:
chunks.append(current_chunk.strip())
current_chunk = ""
current_words = 0
# Split long sentences at clause boundaries, keeping punctuation.
# Only split where whitespace already follows the punctuation so
# tokens are never broken apart or re-joined with added spaces
# (no spaces are injected into "1,000,000" or "12:30").
parts = re.split(r"(?<=[,;:])\s+", sentence)
for part in parts:
part_words = len(part.split())
if part_words > max_words:
# Last resort: no punctuation split point is available,
# so split at word boundaries. Tokens themselves (and
# therefore numbers like "1,000,000") stay intact.
if current_chunk:
chunks.append(current_chunk.strip())
current_chunk = ""
current_words = 0
words = part.split()
for start in range(0, len(words), max_words):
chunks.append(" ".join(words[start:start + max_words]))
continue
if current_words + part_words <= max_words:
current_chunk += part + " "
current_words += part_words
else:
if current_chunk:
chunks.append(current_chunk.strip())
current_chunk = part + " "
current_words = part_words
else:
if current_words + sentence_words <= max_words:
current_chunk += sentence + " "
current_words += sentence_words
else:
if current_chunk:
chunks.append(current_chunk.strip())
current_chunk = sentence + " "
current_words = sentence_words
if current_chunk.strip():
chunks.append(current_chunk.strip())
return [chunk for chunk in chunks if chunk.strip()]
|