aboutsummaryrefslogtreecommitdiff
path: root/converter/chunking.py
blob: 0c85adfb9584264d3d82ff32cf5f95555efc1a9e (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
"""Split extracted book text into TTS-sized chunks."""

import re
from typing import List

from . import config


def split_into_chunks(text: str, max_words: int = config.CHUNK_SIZE_WORDS) -> List[str]:
    """Split text into chunks of at most ``max_words`` words.

    Splits on sentence boundaries. Sentences longer than the limit are split
    further at clause punctuation (which is kept attached for TTS prosody).
    Clause splits only happen at whitespace after punctuation, so tokens like
    "1,000,000" or "12:30" are never broken apart or re-joined with added
    spaces. A single sentence with no usable split point longer than the
    limit is kept intact as one oversized chunk.
    """
    if not text.strip():
        return []

    sentences = re.split(r"(?<=[.!?])\s+", text)
    chunks = []
    current_chunk = ""
    current_words = 0

    for sentence in sentences:
        sentence_words = len(sentence.split())

        if sentence_words > max_words:
            if current_chunk:
                chunks.append(current_chunk.strip())
                current_chunk = ""
                current_words = 0

            # Split long sentences at clause boundaries, keeping punctuation.
            # Only split where whitespace already follows the punctuation so
            # the reassembled text is byte-identical to the input (no spaces
            # injected into "1,000,000" or "12:30").
            parts = re.split(r"(?<=[,;:])\s+", sentence)
            for part in parts:
                part_words = len(part.split())
                if current_words + part_words <= max_words:
                    current_chunk += part + " "
                    current_words += part_words
                else:
                    if current_chunk:
                        chunks.append(current_chunk.strip())
                    current_chunk = part + " "
                    current_words = part_words
        else:
            if current_words + sentence_words <= max_words:
                current_chunk += sentence + " "
                current_words += sentence_words
            else:
                if current_chunk:
                    chunks.append(current_chunk.strip())
                current_chunk = sentence + " "
                current_words = sentence_words

    if current_chunk.strip():
        chunks.append(current_chunk.strip())

    return [chunk for chunk in chunks if chunk.strip()]