diff options
Diffstat (limited to 'app/converter/tts.py')
| -rw-r--r-- | app/converter/tts.py | 1305 |
1 files changed, 1305 insertions, 0 deletions
diff --git a/app/converter/tts.py b/app/converter/tts.py new file mode 100644 index 0000000..6ccef56 --- /dev/null +++ b/app/converter/tts.py @@ -0,0 +1,1305 @@ +"""Client wrappers for the TTS backends. + +QwenTTSClient talks to the Qwen3-TTS demo server (custom voice / voice clone). +FasterTTSClient talks to the OpenAI-compatible server from the +faster-qwen3-tts repository (voice cloning only; the reference voice is +configured server-side — see the "Faster backend" section of the README). +AudioCppTTSClient talks to the audiocpp_server from the audio.cpp +repository, which can host any TTS model family audio.cpp supports +(Qwen3-TTS, Higgs Audio, VoxCPM2, IndexTTS2, ...) through one OpenAI-style +API; the family is detected from the server at startup (see the +"audio.cpp backend" sections of the README). +""" + +import contextlib +import io +import json +import logging +import random +import shutil +import sys +import tempfile +import threading +import time +import urllib.error +import urllib.parse +import urllib.request +import wave +from pathlib import Path +from typing import Any, Dict, List, Optional, Tuple + +from . import config +from .audio import concat_audio_files +from .chunking import split_into_chunks + +logger = logging.getLogger(__name__) + +# Voice modes (re-exported for the CLI and the converter orchestrator). +VOICE_MODE_CUSTOM = "custom_voice" +VOICE_MODE_CLONE = "voice_clone" +VOICE_MODES = (VOICE_MODE_CUSTOM, VOICE_MODE_CLONE) + +# TTS backends (re-exported for the CLI and the converter orchestrator). +BACKEND_QWEN = "qwen" +BACKEND_FASTER = "faster" +BACKEND_AUDIOCPP = "audiocpp" +BACKENDS = (BACKEND_AUDIOCPP, BACKEND_QWEN, BACKEND_FASTER) + +# Languages understood by the Qwen3-TTS API. Display names must match the +# demo dropdown exactly (the demo silently falls back to "Auto" for +# unrecognized values, so languages are validated client-side first). +TTS_LANGUAGES = ( + "Auto", + "Chinese", + "English", + "German", + "Italian", + "Portuguese", + "Spanish", + "Japanese", + "Korean", + "French", + "Russian", +) + +# Short aliases accepted on the command line (ISO 639-1 codes and common +# shorthands), mapped to the display names above. +TTS_LANGUAGE_ALIASES = { + "zh": "Chinese", + "en": "English", + "de": "German", + "it": "Italian", + "pt": "Portuguese", + "es": "Spanish", + "ja": "Japanese", + "ko": "Korean", + "fr": "French", + "ru": "Russian", + "zh-cn": "Chinese", + "zh-tw": "Chinese", + "pt-br": "Portuguese", + "en-us": "English", + "en-gb": "English", +} + +# Qwen display names -> ISO 639-1 codes, for audio.cpp families whose +# language request option takes a code instead of a display name. "Auto" +# has no code and maps to None so the field is omitted and the server +# applies its own default. +LANGUAGE_ISO_CODES = { + "Chinese": "zh", + "English": "en", + "German": "de", + "Italian": "it", + "Portuguese": "pt", + "Spanish": "es", + "Japanese": "ja", + "Korean": "ko", + "French": "fr", + "Russian": "ru", +} + +# --- audio.cpp model families --------------------------------------------- +# +# audiocpp_server exposes the same OpenAI-style API for every TTS family it +# hosts; families only differ in a few request conventions, captured here as +# profiles. Families that are not listed use the default profile below. + +# How the "language" request field is expressed by a family. +AUDIOCPP_LANG_DISPLAY = "display" # Qwen display names, e.g. "English" +AUDIOCPP_LANG_ISO = "iso" # ISO 639-1 codes, e.g. "en" +AUDIOCPP_LANG_OMIT = "omit" # no language field; the model detects it + +# The only family with a built-in speaker mode (CustomVoice speaker names +# plus the INSTRUCT style prompt). Every other family is clone-only: the +# voice comes from a server-side preset requested with --voice. +AUDIOCPP_FAMILY_QWEN3_TTS = "qwen3_tts" + +# Server model entry tasks this client can synthesize audiobooks with, +# taken from GET /v1/models (the "task" field of each entry; servers that +# predate the field reported TTS models only, so a missing task is treated +# as "tts"). "vdes" entries are voice design models: the voice is described +# with --instructions instead of coming from a speaker or a reference clip. +# Entries with any other task (asr, vc, diar, ...) are rejected at connect +# time with a hint to pick a synthesis entry. +AUDIOCPP_TASK_TTS = "tts" +AUDIOCPP_TASK_VDES = "vdes" +AUDIOCPP_SYNTHESIS_TASKS = (AUDIOCPP_TASK_TTS, "clon", AUDIOCPP_TASK_VDES) + + +class AudioCppFamilyProfile: + """Request conventions of one audio.cpp model family.""" + + def __init__(self, language_style: str = AUDIOCPP_LANG_OMIT, + sends_instructions: bool = False, + builtin_speakers: bool = False): + self.language_style = language_style + self.sends_instructions = sends_instructions + self.builtin_speakers = builtin_speakers + + +# Generic profile for families not listed in AUDIOCPP_FAMILY_PROFILES: +# clone-only, no style instructions, and no language field (the model +# detects the language itself). Describes higgs_audio_tts, voxcpm2, +# fish_audio, dots_tts, dramabox, omnivoice, outetts, glm_tts, miotts, +# moss_tts_*, pocket_tts, vibevoice, ... as well as families added to +# audio.cpp after this table was written. +AUDIOCPP_DEFAULT_FAMILY_PROFILE = AudioCppFamilyProfile() + +AUDIOCPP_FAMILY_PROFILES = { + AUDIOCPP_FAMILY_QWEN3_TTS: AudioCppFamilyProfile( + language_style=AUDIOCPP_LANG_DISPLAY, + sends_instructions=True, + builtin_speakers=True, + ), + # Families whose language option takes a code (e.g. "en") instead of + # a Qwen display name; otherwise clone-only like the default profile. + "chatterbox": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO), + "confucius4_tts": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO), + "index_tts2": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO), + "magpie_tts": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO), + "supertonic": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO), +} + +# Canonical speaker names -> display names used by the qwen-tts demo. +SPEAKER_DISPLAY_NAMES = { + "ryan": "Ryan", + "serena": "Serena", + "vivian": "Vivian", + "uncle_fu": "Uncle Fu", + "aiden": "Aiden", + "ono_anna": "Ono Anna", + "sohee": "Sohee", + "eric": "Eric", + "dylan": "Dylan", +} + +# Fixed model facts: both demos run the 1.7B model (the CustomVoice demo +# takes its full HuggingFace id), and the 12Hz codec outputs 24 kHz audio. +MODEL_SIZE = "1.7B" +CUSTOM_VOICE_MODEL_ID = "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice" +SAMPLE_RATE = 24000 + +CHUNKS_FOLDER = Path(__file__).resolve().parent.parent.parent / "app" / "chunks" + + +def _resolve_request_seed() -> int: + """Resolve the seed sent with every request. + + Returns config.SEED as-is, or (with CONSTANT_SEED and SEED < 0) one + random value drawn per run, meant to be reused for every request so + the voice stays consistent across chunk boundaries. Without + CONSTANT_SEED, -1 is returned so the server re-samples the voice on + every generation. + """ + seed = config.SEED + if config.CONSTANT_SEED and seed < 0: + seed = random.randrange(2 ** 31) + return seed + + +def speaker_display_name() -> str: + """Return the display name for the configured custom speaker.""" + return SPEAKER_DISPLAY_NAMES.get( + config.SPEAKER.lower(), config.SPEAKER) + + +def normalize_language(value: Optional[str]) -> str: + """Normalize a user-provided language name to a Qwen3-TTS display name. + + Accepts the display names in TTS_LANGUAGES case-insensitively as + well as the short aliases in TTS_LANGUAGE_ALIASES (ISO 639-1 codes + and common shorthands). Raises ValueError for anything else, since the + Qwen3-TTS demo silently falls back to "Auto" for unrecognized languages. + """ + if value is None: + raise ValueError("Language must not be None") + candidate = value.strip() + if not candidate: + raise ValueError("Language must not be empty") + for name in TTS_LANGUAGES: + if candidate.lower() == name.lower(): + return name + alias = TTS_LANGUAGE_ALIASES.get(candidate.lower()) + if alias: + return alias + raise ValueError( + f"Unknown language: {value!r}. Expected one of " + f"{', '.join(TTS_LANGUAGES)} (or an alias: " + f"{', '.join(sorted(TTS_LANGUAGE_ALIASES))})." + ) + + +def transcribe_reference_audio(audio_path: str, model_name: str = "base") -> Optional[str]: + """Transcribe reference audio locally using an optional Whisper backend. + + The current qwen-tts demo does not expose a transcription endpoint, so + transcription is done client-side when a Whisper package is available. + Returns None if no backend is installed. + """ + for backend in ("faster_whisper", "whisper"): + try: + if backend == "faster_whisper": + from faster_whisper import WhisperModel + model = WhisperModel(model_name, device="cpu", compute_type="int8") + segments, _ = model.transcribe(audio_path) + text = " ".join(seg.text.strip() for seg in segments).strip() + else: + import whisper + model = whisper.load_model(model_name) + result = model.transcribe(audio_path) + text = (result.get("text") or "").strip() + if text: + logger.info("Transcription complete via %s: %s", backend, text) + return text + except ImportError: + continue + except Exception as exc: + logger.warning("%s transcription failed: %s", backend, exc) + logger.warning("No Whisper backend available; transcription skipped.") + return None + + +def whisper_backend_available() -> Optional[str]: + """Return the name of an importable Whisper backend, or None. + + Checks faster_whisper first (preferred), then the openai-whisper + package, without importing the heavy model code: a bare import probe + is enough to tell whether the package is installed in the current + environment. Used by the make_audiocpp_server_json tool to warn when + neither is present (e.g. the wrong conda environment is active). + """ + for backend in ("faster_whisper", "whisper"): + try: + __import__(backend) + except ImportError: + continue + return backend + return None + + +# 150 wpm is a typical spoken pace; used only to size the HTTP request +# timeout for long audio.cpp generations (not as a correctness check). +_ESTIMATED_WORDS_PER_MINUTE = 150 + + +class _BaseTTSClient: + """Shared chunk retry logic, heartbeat, and chunk file bookkeeping.""" + + def generate_chunk(self, text: str, chunk_num: int) -> Optional[str]: + """Generate one audio chunk; returns its path in the chunks folder.""" + raise NotImplementedError + + def _chunk_path(self, chunk_num: int, suffix: str) -> Path: + """Resolve the target path for a chunk, removing stale files first. + + Any stale chunk file for this index is removed so a retry or extension + change can never leave two files matching chunk_NNNN.*. + """ + for stale in CHUNKS_FOLDER.glob(f"chunk_{chunk_num:04d}.*"): + try: + stale.unlink() + except OSError as exc: + logger.debug("Could not remove stale chunk file %s: %s", stale, exc) + return CHUNKS_FOLDER / f"chunk_{chunk_num:04d}{suffix}" + + def process_chunk_with_retry(self, chunk_num: int, text: str) -> Optional[Path]: + """Process a chunk with retry logic. + + Returns the generated chunk file's path, or None when all attempts + failed. + """ + for attempt in range(config.MAX_RETRIES): + try: + result = self.generate_chunk(text, chunk_num) + if result and Path(result).exists(): + return Path(result) + logger.warning("Chunk %d attempt %d failed", chunk_num, attempt + 1) + except Exception as exc: + logger.warning("Chunk %d attempt %d error: %s", chunk_num, attempt + 1, exc) + + if attempt < config.MAX_RETRIES - 1: + sleep_time = 5 + (2 ** attempt) + logger.info("Waiting %ds before retry...", sleep_time) + time.sleep(sleep_time) + + logger.error("Chunk %d failed after %d attempts", chunk_num, config.MAX_RETRIES) + return None + + @contextlib.contextmanager + def _chunk_heartbeat(self, chunk_num: int, label: Optional[str] = None): + """Print a periodic "still working" message while a request generates. + + ``label`` overrides the default "Chunk {chunk_num}" subject, for + backends that send one request per chapter without client-side + chunking (the audio.cpp default) where "chunk" would be misleading. + """ + stop = threading.Event() + subject = label if label is not None else f"Chunk {chunk_num}" + + def _beat(): + start = time.time() + while not stop.wait(config.HEARTBEAT_INTERVAL_SECONDS): + elapsed = time.time() - start + print(f"[...] {subject} still generating — " + f"{int(elapsed // 60)}m {int(elapsed % 60)}s elapsed", flush=True) + + thread = threading.Thread(target=_beat, daemon=True) + thread.start() + try: + yield + finally: + stop.set() + thread.join() + + +class QwenTTSClient(_BaseTTSClient): + """Generates audio chunks through a Qwen3-TTS demo server.""" + + def __init__(self, voice_mode: str = "custom_voice", voice_clone_ref_audio: Optional[str] = None, + voice_clone_ref_text: Optional[str] = None, skip_transcription: bool = False, + language: Optional[str] = None): + if voice_mode not in VOICE_MODES: + raise ValueError( + f"Unknown voice mode: {voice_mode!r} (expected one of {VOICE_MODES})" + ) + self.voice_mode = voice_mode + self.voice_clone_ref_audio = voice_clone_ref_audio + self.voice_clone_ref_text = (voice_clone_ref_text or "").strip() + self.skip_transcription = skip_transcription + # Seed sent with every request: config.SEED as-is, or (with + # CONSTANT_SEED and SEED < 0) one random value drawn per run and + # reused for every request so the voice stays consistent across + # chunk boundaries. Without CONSTANT_SEED, -1 is forwarded so the + # server re-samples the voice on every generation. + self._seed = _resolve_request_seed() + if language is None: + language = config.LANGUAGE + # Validate before connecting so bad values fail fast without a server. + self.language = normalize_language(language) + self.client = None + self.api_info: Dict[str, Any] = {} + self.clone_client = None + self.clone_api_info: Dict[str, Any] = {} + self._ref_audio_filedata: Optional[Dict[str, Any]] = None + self._connect() + + # ------------------------------------------------------------------ + # Connection + # ------------------------------------------------------------------ + + def _connect(self) -> None: + api_url = config.CLONE_API_URL if self.voice_mode == VOICE_MODE_CLONE else config.QWEN_API_URL + try: + if self.voice_mode == VOICE_MODE_CLONE: + # Voice clone uses the Base-model demo, which is a separate server + # from the CustomVoice demo (that one only exposes /run_instruct). + self._init_client(config.CLONE_API_URL, clone=True) + print(f"[OK] Connected to Voice Clone API at {config.CLONE_API_URL}") + self._resolve_reference_text() + else: + self._init_client(config.QWEN_API_URL, clone=False) + print("[OK] Connected to Qwen API") + except Exception as exc: + raise RuntimeError( + f"Qwen API initialization failed at {api_url}: {exc}. " + "Make sure the Qwen demo server is running and reachable, and that your " + "installed Qwen3-TTS version matches this converter's API expectations " + "(voice clone requires the Base-model demo: Qwen/Qwen3-TTS-12Hz-1.7B-Base)." + ) from exc + + def _resolve_reference_text(self) -> None: + """Resolve the reference transcript: explicit text, then local + transcription, then x-vector-only mode.""" + if not self.voice_clone_ref_text and self.voice_clone_ref_audio: + if self.skip_transcription: + print("[INFO] Skipping reference audio transcription (--no-transcription).") + else: + print("[INFO] Transcribing reference audio for voice cloning...") + self.voice_clone_ref_text = self.transcribe_audio(self.voice_clone_ref_audio) or "" + if not self.voice_clone_ref_text: + print("[WARNING] No reference text available; using x-vector-only clone mode (lower quality).") + print(' Pass --transcription "..." for higher-quality in-context cloning.') + else: + print(f"[OK] Reference text:\n{self.voice_clone_ref_text}") + + def _init_client(self, url: str, clone: bool = False) -> None: + """Initialize a Gradio client and store its API metadata. + + gradio_client prints its usage info directly to stdout while the + client is created and its API metadata loaded, so stdout is swapped + for a buffer for the whole process; the captured text is re-emitted + at DEBUG level for troubleshooting. + """ + from gradio_client import Client + + logger.info("Connecting to Qwen API at %s...", url) + old_stdout = sys.stdout + captured = io.StringIO() + sys.stdout = captured + try: + try: + client = Client(url, httpx_kwargs={"timeout": config.API_TIMEOUT}) + except TypeError: + # Older gradio_client versions don't support httpx_kwargs. + client = Client(url) + if clone: + self.clone_client = client + self.clone_api_info = self._load_api_info(client) + else: + self.client = client + self.api_info = self._load_api_info(client) + finally: + sys.stdout = old_stdout + usage_info = captured.getvalue().strip() + if usage_info: + logger.debug("Gradio client output for %s:\n%s", url, usage_info) + logger.info("Connected to Qwen API") + + @staticmethod + def _load_api_info(client) -> Dict[str, Any]: + """Load available API metadata from the Gradio app.""" + try: + return client.view_api(return_format="dict") + except Exception as exc: + logger.warning("Unable to read API metadata: %s", exc) + return {} + + def _resolve_api_name(self, *candidates: str, api_info: Optional[Dict[str, Any]] = None) -> str: + """Return the first available api_name from candidate list.""" + info = api_info if api_info is not None else self.api_info + named_endpoints = info.get("named_endpoints", {}) + for candidate in candidates: + if candidate in named_endpoints: + return candidate + return candidates[0] + + def _endpoint_accepts_param(self, api_name: str, param_name: str, + api_info: Optional[Dict[str, Any]] = None) -> bool: + """Check whether endpoint input schema includes the given parameter.""" + info = api_info if api_info is not None else self.api_info + endpoint = info.get("named_endpoints", {}).get(api_name, {}) + parameters = endpoint.get("parameters", []) + return any(parameter.get("parameter_name") == param_name for parameter in parameters) + + # ------------------------------------------------------------------ + # Reference audio transcription (voice clone) + # ------------------------------------------------------------------ + + def transcribe_audio(self, audio_path: str) -> Optional[str]: + """Transcribe reference audio locally using an optional Whisper backend.""" + return transcribe_reference_audio(audio_path) + + # ------------------------------------------------------------------ + # Chunk generation + # ------------------------------------------------------------------ + + def generate_chunk(self, text: str, chunk_num: int) -> Optional[str]: + """Generate one audio chunk; returns its path in the chunks folder. + + The text is split into sub-requests of at most + ``config.CHUNK_SIZE`` words each (the book-level chunker + normally guarantees this already; the split is defense in depth + against pathological input such as a punctuation-free run of + text), and the audio files returned for the sub-requests are + concatenated into one chunk file. + """ + try: + sub_texts = split_into_chunks(text, max_words=config.CHUNK_SIZE) + if not sub_texts: + raise RuntimeError("No text to synthesize") + + output_path: Optional[Path] = None + with tempfile.TemporaryDirectory(prefix="tts_parts_") as parts_dir, \ + self._chunk_heartbeat(chunk_num): + part_paths = [ + self._generate_sub_request(sub_text, parts_dir, sub_num, + len(sub_texts), chunk_num) + for sub_num, sub_text in enumerate(sub_texts, 1) + ] + if len(part_paths) == 1: + suffix = part_paths[0].suffix or ".wav" + output_path = self._chunk_path(chunk_num, suffix) + shutil.copy2(part_paths[0], output_path) + else: + output_path = self._chunk_path(chunk_num, ".wav") + concat_audio_files(part_paths, output_path) + + logger.debug("Chunk %d generated successfully (%d sub-request(s))", + chunk_num, len(sub_texts)) + return str(output_path) + + except Exception as exc: + logger.error("Qwen chunk processing failed for chunk %d: %s", chunk_num, exc) + return None + + def _generate_sub_request(self, text: str, parts_dir: str, sub_num: int, + sub_total: int, chunk_num: int) -> Path: + """Run one API generation for ``text``; returns the downloaded audio.""" + if sub_total > 1: + logger.info("Chunk %d: oversized input split into %d requests " + "(sub-request %d/%d)", chunk_num, sub_total, sub_num, sub_total) + if self.voice_mode == VOICE_MODE_CUSTOM: + result = self._generate_custom_voice(text) + elif self.voice_mode == VOICE_MODE_CLONE: + result = self._generate_voice_clone(text) + else: + raise ValueError(f"Unknown voice mode: {self.voice_mode}") + + if not isinstance(result, (tuple, list)) or not result: + raise RuntimeError("Qwen API returned an invalid result") + + audio_path = result[0] # First element is the audio file path + if not isinstance(audio_path, (str, Path)) or not audio_path: + raise RuntimeError("Qwen API did not return an audio file path") + + source = Path(audio_path) + if not source.exists(): + raise RuntimeError(f"Generated audio file not found: {audio_path}") + + destination = Path(parts_dir) / f"part_{sub_num:02d}{source.suffix or '.wav'}" + shutil.copy2(source, destination) + + return destination + + # ------------------------------------------------------------------ + # API payloads + # ------------------------------------------------------------------ + + def _generate_custom_voice(self, text: str) -> Tuple: + """Generate audio using CustomVoice mode.""" + custom_api = self._resolve_api_name("/run_instruct", "/run_custom_voice", "/generate_custom_voice") + if custom_api == "/run_instruct": + payload = dict( + text=text, + lang_disp=self.language, + spk_disp=speaker_display_name(), + instruct=config.INSTRUCT, + ) + else: + payload = dict( + text=text, + language=self.language, + speaker=config.SPEAKER, + instruct=config.INSTRUCT, + ) + if self._endpoint_accepts_param(custom_api, "model_id_cv"): + payload["model_id_cv"] = CUSTOM_VOICE_MODEL_ID + elif self._endpoint_accepts_param(custom_api, "model_size"): + payload["model_size"] = MODEL_SIZE + + if self._endpoint_accepts_param(custom_api, "seed"): + payload["seed"] = self._seed + + return self.client.predict(**payload, api_name=custom_api) + + def _ref_audio_payload(self) -> Dict[str, Any]: + """Gradio file payload for the reference audio (built once, reused).""" + if self._ref_audio_filedata is None: + from gradio_client import handle_file + self._ref_audio_filedata = handle_file(self.voice_clone_ref_audio) + return self._ref_audio_filedata + + def _generate_voice_clone(self, text: str) -> Tuple: + """Generate audio using Voice Clone mode.""" + if not Path(self.voice_clone_ref_audio).exists(): + raise FileNotFoundError(f"Reference audio not found: {self.voice_clone_ref_audio}") + + if self.clone_client is None: + raise RuntimeError("Voice Clone client is not initialized. Is the Base-model demo running?") + + clone_api = self._resolve_api_name("/run_voice_clone", "/generate_voice_clone", + api_info=self.clone_api_info) + use_xvector = config.XVECTOR_ONLY or not self.voice_clone_ref_text + + if clone_api == "/run_voice_clone": + payload = dict( + ref_aud=self._ref_audio_payload(), + ref_txt=self.voice_clone_ref_text, + use_xvec=use_xvector, + text=text, + lang_disp=self.language, + ) + else: + payload = dict( + ref_audio=self._ref_audio_payload(), + ref_text=self.voice_clone_ref_text, + target_text=text, + language=self.language, + use_xvector_only=use_xvector, + ) + optional_params = { + "model_size": MODEL_SIZE, + "seed": self._seed, + } + for name, value in optional_params.items(): + if self._endpoint_accepts_param(clone_api, name, api_info=self.clone_api_info): + payload[name] = value + + return self.clone_client.predict(**payload, api_name=clone_api) + + +class FasterTTSClient(_BaseTTSClient): + """Generates audio chunks through a faster-qwen3-tts server. + + Talks to the OpenAI-compatible server shipped in the faster-qwen3-tts + repository (examples/openai_server.py). The reference voice (ref audio, + ref text) and language are configured on the server itself via + --ref-audio/--ref-text or a --voices JSON file; this client only sends + text. Unlike the Qwen demo, the server performs one generation per + request, so long chunks are sub-chunked client-side. + """ + + def __init__(self, voice: Optional[str] = None, api_url: Optional[str] = None): + self.voice = voice or config.FASTER_VOICE + self.api_url = (api_url or config.FASTER_API_URL).rstrip("/") + self._check_health() + + def _check_health(self) -> None: + """Verify the server is reachable and its model is loaded.""" + url = f"{self.api_url}/health" + try: + with urllib.request.urlopen(url, timeout=10) as response: + payload = json.loads(response.read().decode("utf-8")) + except Exception as exc: + raise RuntimeError( + f"Faster TTS server not reachable at {url}: {exc}. " + "Start the faster-qwen3-tts OpenAI-compatible server first " + "(see the 'Faster backend' section of the README)." + ) from exc + if not payload.get("model_loaded"): + raise RuntimeError( + "The faster TTS server is running but its model is not loaded yet; " + "wait for model download and startup to finish, then retry." + ) + print(f"[OK] Connected to faster TTS API at {self.api_url} (voice '{self.voice}')") + print(f"[INFO] The server silently falls back to its first configured voice if " + f"'{self.voice}' is not defined in its voice config (see README).") + + # ------------------------------------------------------------------ + # HTTP requests + # ------------------------------------------------------------------ + + def _request_pcm(self, text: str) -> bytes: + """POST one sub-chunk and return raw 16-bit mono PCM bytes.""" + url = f"{self.api_url}/v1/audio/speech" + payload = json.dumps({ + "model": "tts-1", + "input": text, + "voice": self.voice, + "response_format": "pcm", + }).encode("utf-8") + request = urllib.request.Request( + url, data=payload, headers={"Content-Type": "application/json"}, method="POST") + try: + with urllib.request.urlopen(request, timeout=config.API_TIMEOUT) as response: + pcm = response.read() + except urllib.error.HTTPError as exc: + detail = "" + try: + detail = exc.read().decode("utf-8", errors="replace")[:200] + except Exception: + pass + raise RuntimeError(f"Faster TTS server returned HTTP {exc.code}: {detail}") from exc + except urllib.error.URLError as exc: + raise RuntimeError(f"Faster TTS request failed: {exc.reason}") from exc + if not pcm: + raise RuntimeError("Faster TTS server returned empty audio") + return pcm + + def _request_pcm_with_retry(self, text: str, chunk_num: int, sub_num: int, + sub_total: int) -> bytes: + """Request one sub-chunk, retrying transient failures.""" + for attempt in range(config.MAX_RETRIES): + try: + return self._request_pcm(text) + except Exception as exc: + logger.warning("Chunk %d sub-chunk %d/%d attempt %d failed: %s", + chunk_num, sub_num, sub_total, attempt + 1, exc) + if attempt < config.MAX_RETRIES - 1: + time.sleep(2 + 2 * attempt) + raise RuntimeError(f"Sub-chunk {sub_num}/{sub_total} failed after " + f"{config.MAX_RETRIES} attempts") + + # ------------------------------------------------------------------ + # Chunk generation + # ------------------------------------------------------------------ + + def generate_chunk(self, text: str, chunk_num: int) -> Optional[str]: + """Generate one audio chunk; returns its path in the chunks folder.""" + try: + sub_chunks = split_into_chunks(text, max_words=config.CHUNK_SIZE) + if not sub_chunks: + raise RuntimeError("No text to synthesize") + + pcm_parts: List[bytes] = [] + with self._chunk_heartbeat(chunk_num): + for sub_num, sub_text in enumerate(sub_chunks, 1): + pcm = self._request_pcm_with_retry( + sub_text, chunk_num, sub_num, len(sub_chunks)) + pcm_parts.append(pcm) + + output_path = self._chunk_path(chunk_num, ".wav") + with wave.open(str(output_path), "wb") as wav_file: + wav_file.setnchannels(1) + wav_file.setsampwidth(2) + wav_file.setframerate(SAMPLE_RATE) + wav_file.writeframes(b"".join(pcm_parts)) + + logger.debug("Chunk %d generated (%d sub-chunks)", chunk_num, len(sub_chunks)) + return str(output_path) + + except Exception as exc: + logger.error("Faster chunk processing failed for chunk %d: %s", chunk_num, exc) + return None + + +class AudioCppTTSClient(_BaseTTSClient): + """Generates audio chunks through an audio.cpp audiocpp_server. + + Talks to the OpenAI-style HTTP API of audiocpp_server, which hosts TTS + model families through a native ggml runtime (GGUF weights, no Python + serving stack). The server API is family-agnostic; the family and task + of the configured model entry are read from GET /v1/models at startup + and adapt the request payload (language field style, style instructions) + through AUDIOCPP_FAMILY_PROFILES. Three voice modes, all resolved + server-side from the request's "voice"/"instructions" fields: + + - Speaker mode (no ``voice``): Qwen3-TTS only. A built-in CustomVoice + speaker name (e.g. "Vivian") is passed through, plus the INSTRUCT + style prompt. The server must be configured with the CustomVoice + model for this. Families without built-in speakers reject this mode + with a hint to pass --voice (or --instructions, see below). + - Preset mode (``voice=NAME``): a voice configured on the server + (``voice_presets`` or ``voice_dir`` in its config, e.g. a cloning + reference). The name is validated against GET /v1/audio/voices at + startup because an unresolvable name would silently fall back to + plain TTS on a clone-based model instead of failing. When + AUDIOCPP_CLONE_MODEL_ID names a second server entry of the same + family (typically the Qwen Base model), preset requests are routed + to it. Only the entry actually used needs to exist on the server: a + clone-only (Base) server works for --voice runs, while speaker mode + on such a server fails with a hint to pass --voice. + - Voice design (task "vdes" entries, e.g. Qwen3-TTS VoiceDesign): the + voice is described in natural language through ``instructions``, + which is required and sent with every request (no ``voice`` field). + A constant per-run seed keeps the designed voice consistent across + chunk boundaries. + + ``instructions`` also works on non-design entries, where it acts as a + generic style/delivery instruction (voice control): families that read + it (OmniVoice, Qwen3-TTS CustomVoice, ...) shape the voice or delivery + accordingly, and others ignore it. On instruction-conditioned families + without built-in speakers it may replace --voice entirely (the + instruction defines the voice). Extra request options (``--option + KEY=VALUE``, e.g. emotion, voice_id, speed) are forwarded verbatim in + the request's "options" object, which is the server's generic + pass-through for per-model controls. + + Chunking: the server does its own long-form text chunking for every + family (its ``text_chunk_size`` option, with a per-family default), so + by default each chapter is sent as a single request and the audio + comes back already stitched. With ``chunk_text=True`` (the --chunk CLI + flag), text is instead split client-side into CHUNK_SIZE-word + sub-requests, which may needlessly double-chunk — the warning is + printed by the CLI. + + Each response is a complete WAV file, so sub-request audio is + concatenated with the same lossless path used for the Qwen client. + """ + + def __init__(self, voice: Optional[str] = None, language: Optional[str] = None, + api_url: Optional[str] = None, chunk_text: bool = False, + model_id: Optional[str] = None, + instructions: Optional[str] = None, + request_options: Optional[Dict[str, str]] = None): + self.api_url = (api_url or config.AUDIOCPP_API_URL).rstrip("/") + # Per-run model selection: the --model CLI flag overrides config; an + # empty value is resolved at connect time when the server hosts exactly + # one entry, so multi-model servers don't require editing config.py. + self.model_id = (model_id if model_id is not None + else config.AUDIOCPP_MODEL_ID) or "" + self._model_id_explicit = bool(self.model_id) + # Validate before connecting so bad values fail fast without a server. + self.language = normalize_language( + language if language is not None else config.LANGUAGE) + # One seed value per run, reused for every request (see + # _resolve_request_seed). Unlike the Qwen demo, audio.cpp has no + # negative "randomize" seed, so a negative value means "send no seed + # at all" (see _request_wav) and the server randomizes. + self._seed = _resolve_request_seed() + self.preset_mode = bool(voice) + self.voice = voice or speaker_display_name() + # Style/voice-design instruction sent with every request (the CLI + # --instructions flag overrides AUDIOCPP_INSTRUCTIONS in config.py). + # For task "vdes" entries it describes the voice to design; for other + # families it is a generic style instruction when the model reads one. + self.instructions = (instructions if instructions is not None + else config.AUDIOCPP_INSTRUCTIONS or "").strip() + # Free-form per-request options (--option KEY=VALUE) forwarded in the + # request's "options" object; models ignore keys they don't know. + self.request_options: Dict[str, str] = dict(request_options or {}) + # Both set during _connect once the entry's task is known: design_mode + # for "vdes" entries, instruction_voice when a family without built-in + # speakers gets its voice from the instruction alone (no voice field). + self.design_mode = False + self.instruction_voice = False + # When False (default), each chapter is sent as one request and the + # server does its own long-form chunking (text_chunk_size); when True, + # text is split client-side into CHUNK_SIZE-word sub-requests first. + self.chunk_text = bool(chunk_text) + # Family and task of the selected model entry and the family's request + # profile; all are resolved from GET /v1/models during _connect. + self.family = "" + self.task = AUDIOCPP_TASK_TTS + self.profile = AUDIOCPP_DEFAULT_FAMILY_PROFILE + self._connect() + + # ------------------------------------------------------------------ + # Connection + # ------------------------------------------------------------------ + + def _connect(self) -> None: + """Health-check the server and resolve the model, family, task, and voice. + + Speaker mode is only offered to families with built-in speakers + (Qwen3-TTS); every other family must select a server-side voice + with --voice or describe one with --instructions, so it fails fast + with a hint instead of silently synthesizing with a random default + voice. Voice design entries (task "vdes") require --instructions + and reject --voice. + """ + self._check_health() + models = self._list_models() + self._auto_pick_model_id(models) + if self.preset_mode: + self._select_model(models) + self._require_model_id(models) + self._resolve_family(models) + self._resolve_task(models) + if self.task not in AUDIOCPP_SYNTHESIS_TASKS: + available = ", ".join(model["id"] for model in models) or "none" + raise RuntimeError( + f"The audio.cpp model '{self.model_id}' has task " + f"'{self.task}'; audiobook.py can only synthesize with TTS " + f"model entries (tasks {', '.join(AUDIOCPP_SYNTHESIS_TASKS)}). " + f"Pick a synthesis entry with --model (available: {available})." + ) + if self.design_mode: + if self.preset_mode: + raise RuntimeError( + f"--voice cannot be used with the voice design model " + f"'{self.model_id}': the voice is described by the " + "--instructions text instead (see README).") + if not self.instructions: + raise RuntimeError( + f"The audio.cpp model '{self.model_id}' (family " + f"'{self.family}') is a voice design model: pass a " + "description of the voice to synthesize with, e.g. " + '--instructions "A warm adult female narrator with a ' + 'British accent" (see README).') + print(f"[OK] Connected to audio.cpp server at {self.api_url} " + f"(model '{self.model_id}', family '{self.family}', " + "voice design)") + print(f"[INFO] Designing the voice from: {self.instructions}") + elif self.preset_mode: + self._check_voice() + print(f"[OK] Connected to audio.cpp server at {self.api_url} " + f"(model '{self.model_id}', family '{self.family}', " + f"voice '{self.voice}')") + elif self.profile.builtin_speakers: + print(f"[OK] Connected to audio.cpp server at {self.api_url} " + f"(model '{self.model_id}', family '{self.family}', " + f"speaker '{self.voice}')") + print("[INFO] Speaker mode expects the server to be configured with the " + "CustomVoice model; with the Base model the speaker name is ignored " + "and a random default voice is used (see README).") + elif self.instructions: + # Families without built-in speakers can still get their voice + # from the instruction alone (e.g. OmniVoice voice design). + self.instruction_voice = True + print(f"[OK] Connected to audio.cpp server at {self.api_url} " + f"(model '{self.model_id}', family '{self.family}', " + "instruction voice)") + print(f"[INFO] Designing the voice from: {self.instructions}") + else: + raise RuntimeError( + f"The audio.cpp model '{self.model_id}' (family " + f"'{self.family}') has no built-in speakers, so its voice " + "must come from the server: rerun with --voice NAME " + "matching a voice_preset or voice_dir entry in the server " + "config, or describe a voice with --instructions for " + "families that support it (see README).") + if self.instructions and not self.design_mode and not self.instruction_voice: + print(f"[INFO] Sending instruction with every request: {self.instructions}") + print("[INFO] Its effect (style, emotion, delivery) depends on the " + "model family; models without instruction support ignore it.") + + def _get_json(self, path: str, timeout: int = 10) -> Dict[str, Any]: + """GET a JSON document from the server.""" + url = f"{self.api_url}{path}" + try: + with urllib.request.urlopen(url, timeout=timeout) as response: + return json.loads(response.read().decode("utf-8")) + except urllib.error.HTTPError as exc: + detail = "" + try: + detail = exc.read().decode("utf-8", errors="replace")[:200] + except Exception: + pass + raise RuntimeError( + f"audio.cpp server returned HTTP {exc.code} for {path}: {detail}") from exc + except urllib.error.URLError as exc: + raise RuntimeError(f"audio.cpp request failed for {path}: {exc.reason}") from exc + + def _check_health(self) -> None: + """Verify the server is reachable and reports healthy.""" + try: + payload = self._get_json("/health") + except Exception as exc: + raise RuntimeError( + f"audio.cpp server not reachable at {self.api_url}: {exc}. " + "Start audiocpp_server first (see the 'audio.cpp backend' " + "section of the README)." + ) from exc + if payload.get("status") != "ok": + raise RuntimeError( + f"The audio.cpp server at {self.api_url} reports status " + f"{payload.get('status')!r} instead of 'ok'") + + def _list_models(self) -> List[Dict[str, str]]: + """Fetch the (id, family, task) triples reported by the server.""" + try: + payload = self._get_json("/v1/models") + except Exception as exc: + raise RuntimeError( + f"The audio.cpp server at {self.api_url} did not answer " + f"/v1/models: {exc}") from exc + entries = payload.get("data") or [] + models: List[Dict[str, str]] = [] + for entry in entries: + if isinstance(entry, dict) and entry.get("id"): + models.append({ + "id": entry["id"], + "family": entry.get("family") or "", + "task": entry.get("task") or "", + }) + return models + + def _auto_pick_model_id(self, models: List[Dict[str, str]]) -> None: + """Resolve an empty model id when the server hosts exactly one entry. + + Multi-model servers generated with several lazily-loaded entries can + be used without editing app/converter/config.py: leave AUDIOCPP_MODEL_ID + (and ``--model``) unset, and the single hosted entry is chosen + automatically. With more than one entry an explicit choice is required + (via ``--model`` or AUDIOCPP_MODEL_ID), since guessing would risk + synthesizing a whole book with the wrong family. + """ + if self.model_id: + return + if len(models) == 1: + self.model_id = models[0]["id"] + logger.info( + "AUDIOCPP_MODEL_ID is unset; using the only server entry '%s'", + self.model_id) + else: + logger.debug( + "AUDIOCPP_MODEL_ID is unset and the server hosts %d entries; " + "an explicit --model or config id is required", + len(models)) + + def _require_model_id(self, models: List[Dict[str, str]]) -> None: + """Verify the model id chosen for this run exists on the server. + + Speaker mode needs AUDIOCPP_MODEL_ID (the CustomVoice entry). + Preset mode validates whichever id _select_model resolved, so a + server hosting only a cloning model works for --voice. + """ + model_ids = [model["id"] for model in models] + if self.model_id and self.model_id in model_ids: + return + configured = ", ".join(model_ids) or "none" + if not self.model_id: + raise RuntimeError( + f"The audio.cpp server at {self.api_url} hosts {len(model_ids)} " + f"model entries ({configured}); audiobook.py needs to know which " + "one to use. Pass --model <id> when converting, or set " + "AUDIOCPP_MODEL_ID in app/converter/config.py to one of them " + "(see README)." + ) + if self.preset_mode: + raise RuntimeError( + f"The audio.cpp server at {self.api_url} has no model id " + f"'{self.model_id}' or clone model id " + f"'{config.AUDIOCPP_CLONE_MODEL_ID}' (configured: {configured}). " + "Add a TTS model entry for the family you want to the server " + "config and match AUDIOCPP_MODEL_ID / AUDIOCPP_CLONE_MODEL_ID " + "in app/converter/config.py to its id, or select it per run with " + "--model (see README)." + ) + raise RuntimeError( + f"The audio.cpp server at {self.api_url} has no model id " + f"'{self.model_id}' (configured: {configured}). Speaker mode needs " + "the Qwen3-TTS CustomVoice model: add a qwen3_tts model entry to " + "the server config and match AUDIOCPP_MODEL_ID in app/converter/config.py to its " + "id (or pass --model), or rerun with --voice to use a voice preset " + "on any TTS model (see README)." + ) + + def _select_model(self, models: List[Dict[str, str]]) -> None: + """Pick the model for preset (cloning) requests. + + Defaults to the primary model id. When AUDIOCPP_CLONE_MODEL_ID is + configured and present on the server, preset requests are routed + to it instead, so one server can host the CustomVoice model for + speaker mode and the Base model for cloning (Qwen3-TTS setups). + A clone id that names a model of a different family is ignored + with a warning, since preset requests must synthesize with the + family the run is configured for. + """ + clone_model_id = config.AUDIOCPP_CLONE_MODEL_ID + if not clone_model_id or clone_model_id == self.model_id: + return + families = {model["id"]: model["family"] for model in models} + if clone_model_id not in families: + # A qwen3_tts primary without its clone entry silently degrades + # (presets are ignored on the CustomVoice model), so that case + # keeps the warning; single-model servers of other families are + # the normal configuration and only get a debug note. + primary_is_qwen = (families.get(self.model_id) + or AUDIOCPP_FAMILY_QWEN3_TTS) \ + == AUDIOCPP_FAMILY_QWEN3_TTS + if primary_is_qwen: + logger.warning( + "AUDIOCPP_CLONE_MODEL_ID %r is not configured on the audio.cpp " + "server; preset requests use '%s' instead", + clone_model_id, self.model_id) + else: + logger.debug( + "AUDIOCPP_CLONE_MODEL_ID %r is not configured on the audio.cpp " + "server; preset requests use '%s' instead", + clone_model_id, self.model_id) + return + primary_family = families.get(self.model_id) + clone_family = families[clone_model_id] + if primary_family and clone_family and primary_family != clone_family: + logger.warning( + "AUDIOCPP_CLONE_MODEL_ID %r hosts family %r, but " + "AUDIOCPP_MODEL_ID %r hosts %r; preset requests stay on " + "'%s'. Point both ids at the same model entry in " + "app/converter/config.py (single-model servers use the same id " + "for both)", + clone_model_id, clone_family, self.model_id, primary_family, + self.model_id) + return + self.model_id = clone_model_id + + def _resolve_family(self, models: List[Dict[str, str]]) -> None: + """Resolve the selected model's family and its request profile. + + The family comes from GET /v1/models. Servers that predate the + family field served Qwen3-TTS only, so a missing family is treated + as qwen3_tts, which also preserves this client's legacy behavior + against those versions. + """ + entry = next( + (model for model in models if model["id"] == self.model_id), None) + family = (entry["family"] if entry is not None else "") or "" + if not family: + family = AUDIOCPP_FAMILY_QWEN3_TTS + logger.debug("Model '%s' reported no family; assuming qwen3_tts", + self.model_id) + self.family = family + self.profile = AUDIOCPP_FAMILY_PROFILES.get( + family, AUDIOCPP_DEFAULT_FAMILY_PROFILE) + if family not in AUDIOCPP_FAMILY_PROFILES: + logger.info( + "audio.cpp family '%s' has no dedicated profile; using the " + "generic profile (voice cloning via --voice, model-detected " + "language)", family) + + def _resolve_task(self, models: List[Dict[str, str]]) -> None: + """Resolve the selected model's task (tts, clon, vdes, ...) and set + design mode for voice design entries. + + The task comes from GET /v1/models and is fixed per server entry by + its server.json config (a VoiceDesign model must be hosted with + "task": "vdes"). Servers that predate the task field hosted plain + TTS models, so a missing task is treated as tts. + """ + entry = next( + (model for model in models if model["id"] == self.model_id), None) + task = (entry["task"] if entry is not None else "") or "" + if not task: + task = AUDIOCPP_TASK_TTS + logger.debug("Model '%s' reported no task; assuming tts", + self.model_id) + self.task = task + self.design_mode = task == AUDIOCPP_TASK_VDES + + def _check_voice(self) -> None: + """Verify the requested voice is available on the server. + + A voice name that matches no server preset or voice-library wav + would be passed through to the model as a cached voice id; on the + Base (cloning) model that is silently ignored and plain TTS audio + comes back, so preset names are validated up front. When the + voices endpoint cannot be queried, validation is skipped with a + warning rather than blocking the run. + """ + query = urllib.parse.urlencode({"model": self.model_id}) + try: + payload = self._get_json(f"/v1/audio/voices?{query}") + except Exception as exc: + logger.warning("Could not list server voices; skipping voice " + "validation: %s", exc) + return + voices = payload.get("voices") or [] + if self.voice not in voices: + available = ", ".join(str(v) for v in voices) or "none" + raise RuntimeError( + f"Voice '{self.voice}' is not available on the audio.cpp server " + f"(available: {available}). Configure it as a voice_preset or " + "voice_dir entry in the server config, or pass a listed name " + "with --voice (see README)." + ) + + # ------------------------------------------------------------------ + # HTTP requests + # ------------------------------------------------------------------ + + def _request_wav(self, text: str) -> bytes: + """POST one sub-chunk and return the raw WAV bytes. + + The request timeout scales with the text length when a whole + chapter is sent in one request (no client-side chunking), since a + long chapter means many minutes of audio generated in one go. + """ + url = f"{self.api_url}/v1/audio/speech" + payload: Dict[str, Any] = { + "model": self.model_id, + "input": text, + } + # Design models take no voice field (the voice comes from the + # instruction); instruction-voice runs on families without built-in + # speakers omit it too, since no speaker or preset was requested. + if not self.design_mode and not self.instruction_voice: + payload["voice"] = self.voice + if self.profile.language_style == AUDIOCPP_LANG_DISPLAY: + payload["language"] = self.language + elif self.profile.language_style == AUDIOCPP_LANG_ISO: + iso_code = LANGUAGE_ISO_CODES.get(self.language) + if iso_code: + payload["language"] = iso_code + else: + # "Auto": no code to send, so let the server pick its default. + logger.debug("%s: no language code for %r; omitted from request", + self.family, self.language) + if self._seed >= 0: + # audio.cpp has no negative "randomize" seed; a negative seed + # means "let the server randomize", so the field is omitted. + payload["seed"] = self._seed + if self.instructions: + # Explicit voice-design or style instruction (required for task + # "vdes" entries; a Ctrl/style control on families that read it). + payload["instructions"] = self.instructions + elif not self.preset_mode and config.INSTRUCT \ + and self.profile.sends_instructions: + # Style instruction for the Qwen3-TTS CustomVoice speakers; + # ignored by the Base (cloning) model and other families. + payload["instructions"] = config.INSTRUCT + if self.request_options: + # Generic per-model controls (--option KEY=VALUE): forwarded + # verbatim; the model ignores keys it does not know. + payload["options"] = dict(self.request_options) + request = urllib.request.Request( + url, data=json.dumps(payload).encode("utf-8"), + headers={"Content-Type": "application/json"}, method="POST") + timeout = config.API_TIMEOUT + if not self.chunk_text: + # Estimated audio duration at 150 wpm, doubled plus a minute of + # slack, bounded below by the configured per-request timeout. + estimated_seconds = 60.0 * len(text.split()) / _ESTIMATED_WORDS_PER_MINUTE + timeout = max(timeout, int(estimated_seconds * 2) + 60) + try: + with urllib.request.urlopen(request, timeout=timeout) as response: + wav = response.read() + except urllib.error.HTTPError as exc: + detail = "" + try: + detail = exc.read().decode("utf-8", errors="replace")[:200] + except Exception: + pass + raise RuntimeError(f"audio.cpp server returned HTTP {exc.code}: {detail}") from exc + except urllib.error.URLError as exc: + raise RuntimeError(f"audio.cpp request failed: {exc.reason}") from exc + if len(wav) < 12 or wav[:4] != b"RIFF" or wav[8:12] != b"WAVE": + raise RuntimeError("audio.cpp server returned audio that is not a WAV file") + return wav + + def _request_wav_with_retry(self, text: str, chunk_num: int, sub_num: int, + sub_total: int) -> bytes: + """Request one sub-chunk, retrying transient failures.""" + for attempt in range(config.MAX_RETRIES): + try: + return self._request_wav(text) + except Exception as exc: + logger.warning("Chunk %d sub-chunk %d/%d attempt %d failed: %s", + chunk_num, sub_num, sub_total, attempt + 1, exc) + if attempt < config.MAX_RETRIES - 1: + time.sleep(2 + 2 * attempt) + raise RuntimeError(f"Sub-chunk {sub_num}/{sub_total} failed after " + f"{config.MAX_RETRIES} attempts") + + # ------------------------------------------------------------------ + # Chunk generation + # ------------------------------------------------------------------ + + def generate_chunk(self, text: str, chunk_num: int) -> Optional[str]: + """Generate one audio chunk; returns its path in the chunks folder. + + By default the whole text goes out as a single request and the + server does its own long-form chunking (see the class docstring). + With ``chunk_text=True`` (--chunk), the text is split into + sub-requests of at most ``config.CHUNK_SIZE`` words each; each + sub-request returns a complete WAV file and the parts are + concatenated into one chunk file. + """ + try: + if self.chunk_text: + sub_texts = split_into_chunks(text, max_words=config.CHUNK_SIZE) + elif text.strip(): + sub_texts = [text] + else: + sub_texts = [] + if not sub_texts: + raise RuntimeError("No text to synthesize") + + output_path: Optional[Path] = None + with tempfile.TemporaryDirectory(prefix="tts_parts_") as parts_dir, \ + self._chunk_heartbeat( + chunk_num, + label=None if self.chunk_text else "Request"): + part_paths = [] + for sub_num, sub_text in enumerate(sub_texts, 1): + wav = self._request_wav_with_retry( + sub_text, chunk_num, sub_num, len(sub_texts)) + destination = Path(parts_dir) / f"part_{sub_num:02d}.wav" + destination.write_bytes(wav) + part_paths.append(destination) + if len(part_paths) == 1: + output_path = self._chunk_path(chunk_num, ".wav") + shutil.copy2(part_paths[0], output_path) + else: + output_path = self._chunk_path(chunk_num, ".wav") + concat_audio_files(part_paths, output_path) + + logger.debug("Chunk %d generated successfully (%d sub-request(s))", + chunk_num, len(sub_texts)) + return str(output_path) + + except Exception as exc: + logger.error("audio.cpp chunk processing failed for chunk %d: %s", + chunk_num, exc) + return None |
