aboutsummaryrefslogtreecommitdiff
path: root/converter/chunking.py
blob: 425765fe0d35c50591a1901f68789bc20d1b8d9d (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
"""Split extracted book text into TTS-sized chunks."""

import logging
import re
from typing import List

from . import config

logger = logging.getLogger(__name__)


def split_into_chunks(text: str, max_words: int = config.CHUNK_SIZE_WORDS) -> List[str]:
    """Split text into chunks of at most ``max_words`` words.

    ``max_words`` is clamped to ``config.MAX_REQUEST_WORDS``: requests
    beyond that ceiling are silently truncated by the TTS servers (no
    error is reported), so chunks larger than the ceiling are never
    produced regardless of configuration.

    Splits on sentence boundaries. Sentences longer than the limit are
    split further at clause punctuation (which is kept attached for TTS
    prosody). Clause splits only happen at whitespace after punctuation,
    so tokens like "1,000,000" or "12:30" are never broken apart. A piece
    with no usable punctuation split point longer than the limit is split
    at word boundaries as a last resort: individual tokens stay intact,
    but whitespace between them is normalized.
    """
    if max_words > config.MAX_REQUEST_WORDS:
        logger.warning(
            "Requested chunk size of %d words exceeds the %d-word request ceiling; "
            "larger requests are silently truncated by the TTS servers, so the "
            "size is clamped to %d words (see MAX_REQUEST_WORDS in converter/config.py)",
            max_words, config.MAX_REQUEST_WORDS, config.MAX_REQUEST_WORDS)
        max_words = config.MAX_REQUEST_WORDS
    if max_words < 1:
        max_words = 1

    if not text.strip():
        return []

    sentences = re.split(r"(?<=[.!?])\s+", text)
    chunks = []
    current_chunk = ""
    current_words = 0

    for sentence in sentences:
        sentence_words = len(sentence.split())

        if sentence_words > max_words:
            if current_chunk:
                chunks.append(current_chunk.strip())
                current_chunk = ""
                current_words = 0

            # Split long sentences at clause boundaries, keeping punctuation.
            # Only split where whitespace already follows the punctuation so
            # tokens are never broken apart or re-joined with added spaces
            # (no spaces are injected into "1,000,000" or "12:30").
            parts = re.split(r"(?<=[,;:])\s+", sentence)
            for part in parts:
                part_words = len(part.split())
                if part_words > max_words:
                    # Last resort: no punctuation split point is available,
                    # so split at word boundaries. Tokens themselves (and
                    # therefore numbers like "1,000,000") stay intact.
                    if current_chunk:
                        chunks.append(current_chunk.strip())
                        current_chunk = ""
                        current_words = 0
                    words = part.split()
                    for start in range(0, len(words), max_words):
                        chunks.append(" ".join(words[start:start + max_words]))
                    continue
                if current_words + part_words <= max_words:
                    current_chunk += part + " "
                    current_words += part_words
                else:
                    if current_chunk:
                        chunks.append(current_chunk.strip())
                    current_chunk = part + " "
                    current_words = part_words
        else:
            if current_words + sentence_words <= max_words:
                current_chunk += sentence + " "
                current_words += sentence_words
            else:
                if current_chunk:
                    chunks.append(current_chunk.strip())
                current_chunk = sentence + " "
                current_words = sentence_words

    if current_chunk.strip():
        chunks.append(current_chunk.strip())

    return [chunk for chunk in chunks if chunk.strip()]