"""Client wrappers for the TTS backends. QwenTTSClient talks to the Qwen3-TTS demo server (custom voice / voice clone). FasterTTSClient talks to the OpenAI-compatible server from the faster-qwen3-tts repository (voice cloning only; the reference voice is configured server-side — see the "Faster backend" section of the README). AudioCppTTSClient talks to the audiocpp_server from the audio.cpp repository, which can host any TTS model family audio.cpp supports (Qwen3-TTS, Higgs Audio, VoxCPM2, IndexTTS2, ...) through one OpenAI-style API; the family is detected from the server at startup (see the "audio.cpp backend" sections of the README). """ import contextlib import io import json import logging import random import shutil import sys import tempfile import threading import time import urllib.error import urllib.parse import urllib.request import wave from pathlib import Path from typing import Any, Dict, List, Optional, Tuple from . import config from .audio import concat_audio_files from .chunking import split_into_chunks logger = logging.getLogger(__name__) # Voice modes (re-exported for the CLI and the converter orchestrator). VOICE_MODE_CUSTOM = "custom_voice" VOICE_MODE_CLONE = "voice_clone" VOICE_MODES = (VOICE_MODE_CUSTOM, VOICE_MODE_CLONE) # TTS backends (re-exported for the CLI and the converter orchestrator). BACKEND_QWEN = "qwen" BACKEND_FASTER = "faster" BACKEND_AUDIOCPP = "audiocpp" BACKENDS = (BACKEND_AUDIOCPP, BACKEND_QWEN, BACKEND_FASTER) # Languages understood by the Qwen3-TTS API. Display names must match the # demo dropdown exactly (the demo silently falls back to "Auto" for # unrecognized values, so languages are validated client-side first). TTS_LANGUAGES = ( "Auto", "Chinese", "English", "German", "Italian", "Portuguese", "Spanish", "Japanese", "Korean", "French", "Russian", ) # Short aliases accepted on the command line (ISO 639-1 codes and common # shorthands), mapped to the display names above. TTS_LANGUAGE_ALIASES = { "zh": "Chinese", "en": "English", "de": "German", "it": "Italian", "pt": "Portuguese", "es": "Spanish", "ja": "Japanese", "ko": "Korean", "fr": "French", "ru": "Russian", "zh-cn": "Chinese", "zh-tw": "Chinese", "pt-br": "Portuguese", "en-us": "English", "en-gb": "English", } # Qwen display names -> ISO 639-1 codes, for audio.cpp families whose # language request option takes a code instead of a display name. "Auto" # has no code and maps to None so the field is omitted and the server # applies its own default. LANGUAGE_ISO_CODES = { "Chinese": "zh", "English": "en", "German": "de", "Italian": "it", "Portuguese": "pt", "Spanish": "es", "Japanese": "ja", "Korean": "ko", "French": "fr", "Russian": "ru", } # --- audio.cpp model families --------------------------------------------- # # audiocpp_server exposes the same OpenAI-style API for every TTS family it # hosts; families only differ in a few request conventions, captured here as # profiles. Families that are not listed use the default profile below. # How the "language" request field is expressed by a family. AUDIOCPP_LANG_DISPLAY = "display" # Qwen display names, e.g. "English" AUDIOCPP_LANG_ISO = "iso" # ISO 639-1 codes, e.g. "en" AUDIOCPP_LANG_OMIT = "omit" # no language field; the model detects it # The only family with a built-in speaker mode (CustomVoice speaker names # plus the INSTRUCT style prompt). Every other family is clone-only: the # voice comes from a server-side preset requested with --voice. AUDIOCPP_FAMILY_QWEN3_TTS = "qwen3_tts" # Server model entry tasks this client can synthesize audiobooks with, # taken from GET /v1/models (the "task" field of each entry; servers that # predate the field reported TTS models only, so a missing task is treated # as "tts"). "vdes" entries are voice design models: the voice is described # with --instructions instead of coming from a speaker or a reference clip. # Entries with any other task (asr, vc, diar, ...) are rejected at connect # time with a hint to pick a synthesis entry. AUDIOCPP_TASK_TTS = "tts" AUDIOCPP_TASK_VDES = "vdes" AUDIOCPP_SYNTHESIS_TASKS = (AUDIOCPP_TASK_TTS, "clon", AUDIOCPP_TASK_VDES) class AudioCppFamilyProfile: """Request conventions of one audio.cpp model family.""" def __init__(self, language_style: str = AUDIOCPP_LANG_OMIT, sends_instructions: bool = False, builtin_speakers: bool = False): self.language_style = language_style self.sends_instructions = sends_instructions self.builtin_speakers = builtin_speakers # Generic profile for families not listed in AUDIOCPP_FAMILY_PROFILES: # clone-only, no style instructions, and no language field (the model # detects the language itself). Describes higgs_audio_tts, voxcpm2, # fish_audio, dots_tts, dramabox, omnivoice, outetts, glm_tts, miotts, # moss_tts_*, pocket_tts, vibevoice, ... as well as families added to # audio.cpp after this table was written. AUDIOCPP_DEFAULT_FAMILY_PROFILE = AudioCppFamilyProfile() AUDIOCPP_FAMILY_PROFILES = { AUDIOCPP_FAMILY_QWEN3_TTS: AudioCppFamilyProfile( language_style=AUDIOCPP_LANG_DISPLAY, sends_instructions=True, builtin_speakers=True, ), # Families whose language option takes a code (e.g. "en") instead of # a Qwen display name; otherwise clone-only like the default profile. "chatterbox": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO), "confucius4_tts": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO), "index_tts2": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO), "magpie_tts": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO), "supertonic": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO), } # Canonical speaker names -> display names used by the qwen-tts demo. SPEAKER_DISPLAY_NAMES = { "ryan": "Ryan", "serena": "Serena", "vivian": "Vivian", "uncle_fu": "Uncle Fu", "aiden": "Aiden", "ono_anna": "Ono Anna", "sohee": "Sohee", "eric": "Eric", "dylan": "Dylan", } # Fixed model facts: both demos run the 1.7B model (the CustomVoice demo # takes its full HuggingFace id), and the 12Hz codec outputs 24 kHz audio. MODEL_SIZE = "1.7B" CUSTOM_VOICE_MODEL_ID = "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice" SAMPLE_RATE = 24000 CHUNKS_FOLDER = Path(__file__).resolve().parent.parent / "chunks" def _resolve_request_seed() -> int: """Resolve the seed sent with every request. Returns config.SEED as-is, or (with CONSTANT_SEED and SEED < 0) one random value drawn per run, meant to be reused for every request so the voice stays consistent across chunk boundaries. Without CONSTANT_SEED, -1 is returned so the server re-samples the voice on every generation. """ seed = config.SEED if config.CONSTANT_SEED and seed < 0: seed = random.randrange(2 ** 31) return seed def speaker_display_name() -> str: """Return the display name for the configured custom speaker.""" return SPEAKER_DISPLAY_NAMES.get( config.SPEAKER.lower(), config.SPEAKER) def normalize_language(value: Optional[str]) -> str: """Normalize a user-provided language name to a Qwen3-TTS display name. Accepts the display names in TTS_LANGUAGES case-insensitively as well as the short aliases in TTS_LANGUAGE_ALIASES (ISO 639-1 codes and common shorthands). Raises ValueError for anything else, since the Qwen3-TTS demo silently falls back to "Auto" for unrecognized languages. """ if value is None: raise ValueError("Language must not be None") candidate = value.strip() if not candidate: raise ValueError("Language must not be empty") for name in TTS_LANGUAGES: if candidate.lower() == name.lower(): return name alias = TTS_LANGUAGE_ALIASES.get(candidate.lower()) if alias: return alias raise ValueError( f"Unknown language: {value!r}. Expected one of " f"{', '.join(TTS_LANGUAGES)} (or an alias: " f"{', '.join(sorted(TTS_LANGUAGE_ALIASES))})." ) def transcribe_reference_audio(audio_path: str, model_name: str = "base") -> Optional[str]: """Transcribe reference audio locally using an optional Whisper backend. The current qwen-tts demo does not expose a transcription endpoint, so transcription is done client-side when a Whisper package is available. Returns None if no backend is installed. """ for backend in ("faster_whisper", "whisper"): try: if backend == "faster_whisper": from faster_whisper import WhisperModel model = WhisperModel(model_name, device="cpu", compute_type="int8") segments, _ = model.transcribe(audio_path) text = " ".join(seg.text.strip() for seg in segments).strip() else: import whisper model = whisper.load_model(model_name) result = model.transcribe(audio_path) text = (result.get("text") or "").strip() if text: logger.info("Transcription complete via %s: %s", backend, text) return text except ImportError: continue except Exception as exc: logger.warning("%s transcription failed: %s", backend, exc) logger.warning("No Whisper backend available; transcription skipped.") return None def whisper_backend_available() -> Optional[str]: """Return the name of an importable Whisper backend, or None. Checks faster_whisper first (preferred), then the openai-whisper package, without importing the heavy model code: a bare import probe is enough to tell whether the package is installed in the current environment. Used by the make_audiocpp_server_json tool to warn when neither is present (e.g. the wrong conda environment is active). """ for backend in ("faster_whisper", "whisper"): try: __import__(backend) except ImportError: continue return backend return None # 150 wpm is a typical spoken pace; used only to size the HTTP request # timeout for long audio.cpp generations (not as a correctness check). _ESTIMATED_WORDS_PER_MINUTE = 150 class _BaseTTSClient: """Shared chunk retry logic, heartbeat, and chunk file bookkeeping.""" def generate_chunk(self, text: str, chunk_num: int) -> Optional[str]: """Generate one audio chunk; returns its path in the chunks folder.""" raise NotImplementedError def _chunk_path(self, chunk_num: int, suffix: str) -> Path: """Resolve the target path for a chunk, removing stale files first. Any stale chunk file for this index is removed so a retry or extension change can never leave two files matching chunk_NNNN.*. """ for stale in CHUNKS_FOLDER.glob(f"chunk_{chunk_num:04d}.*"): try: stale.unlink() except OSError as exc: logger.debug("Could not remove stale chunk file %s: %s", stale, exc) return CHUNKS_FOLDER / f"chunk_{chunk_num:04d}{suffix}" def process_chunk_with_retry(self, chunk_num: int, text: str) -> Optional[Path]: """Process a chunk with retry logic. Returns the generated chunk file's path, or None when all attempts failed. """ for attempt in range(config.MAX_RETRIES): try: result = self.generate_chunk(text, chunk_num) if result and Path(result).exists(): return Path(result) logger.warning("Chunk %d attempt %d failed", chunk_num, attempt + 1) except Exception as exc: logger.warning("Chunk %d attempt %d error: %s", chunk_num, attempt + 1, exc) if attempt < config.MAX_RETRIES - 1: sleep_time = 5 + (2 ** attempt) logger.info("Waiting %ds before retry...", sleep_time) time.sleep(sleep_time) logger.error("Chunk %d failed after %d attempts", chunk_num, config.MAX_RETRIES) return None @contextlib.contextmanager def _chunk_heartbeat(self, chunk_num: int, label: Optional[str] = None): """Print a periodic "still working" message while a request generates. ``label`` overrides the default "Chunk {chunk_num}" subject, for backends that send one request per chapter without client-side chunking (the audio.cpp default) where "chunk" would be misleading. """ stop = threading.Event() subject = label if label is not None else f"Chunk {chunk_num}" def _beat(): start = time.time() while not stop.wait(config.HEARTBEAT_INTERVAL_SECONDS): elapsed = time.time() - start print(f"[...] {subject} still generating — " f"{int(elapsed // 60)}m {int(elapsed % 60)}s elapsed", flush=True) thread = threading.Thread(target=_beat, daemon=True) thread.start() try: yield finally: stop.set() thread.join() class QwenTTSClient(_BaseTTSClient): """Generates audio chunks through a Qwen3-TTS demo server.""" def __init__(self, voice_mode: str = "custom_voice", voice_clone_ref_audio: Optional[str] = None, voice_clone_ref_text: Optional[str] = None, skip_transcription: bool = False, language: Optional[str] = None): if voice_mode not in VOICE_MODES: raise ValueError( f"Unknown voice mode: {voice_mode!r} (expected one of {VOICE_MODES})" ) self.voice_mode = voice_mode self.voice_clone_ref_audio = voice_clone_ref_audio self.voice_clone_ref_text = (voice_clone_ref_text or "").strip() self.skip_transcription = skip_transcription # Seed sent with every request: config.SEED as-is, or (with # CONSTANT_SEED and SEED < 0) one random value drawn per run and # reused for every request so the voice stays consistent across # chunk boundaries. Without CONSTANT_SEED, -1 is forwarded so the # server re-samples the voice on every generation. self._seed = _resolve_request_seed() if language is None: language = config.LANGUAGE # Validate before connecting so bad values fail fast without a server. self.language = normalize_language(language) self.client = None self.api_info: Dict[str, Any] = {} self.clone_client = None self.clone_api_info: Dict[str, Any] = {} self._ref_audio_filedata: Optional[Dict[str, Any]] = None self._connect() # ------------------------------------------------------------------ # Connection # ------------------------------------------------------------------ def _connect(self) -> None: api_url = config.CLONE_API_URL if self.voice_mode == VOICE_MODE_CLONE else config.QWEN_API_URL try: if self.voice_mode == VOICE_MODE_CLONE: # Voice clone uses the Base-model demo, which is a separate server # from the CustomVoice demo (that one only exposes /run_instruct). self._init_client(config.CLONE_API_URL, clone=True) print(f"[OK] Connected to Voice Clone API at {config.CLONE_API_URL}") self._resolve_reference_text() else: self._init_client(config.QWEN_API_URL, clone=False) print("[OK] Connected to Qwen API") except Exception as exc: raise RuntimeError( f"Qwen API initialization failed at {api_url}: {exc}. " "Make sure the Qwen demo server is running and reachable, and that your " "installed Qwen3-TTS version matches this converter's API expectations " "(voice clone requires the Base-model demo: Qwen/Qwen3-TTS-12Hz-1.7B-Base)." ) from exc def _resolve_reference_text(self) -> None: """Resolve the reference transcript: explicit text, then local transcription, then x-vector-only mode.""" if not self.voice_clone_ref_text and self.voice_clone_ref_audio: if self.skip_transcription: print("[INFO] Skipping reference audio transcription (--no-transcription).") else: print("[INFO] Transcribing reference audio for voice cloning...") self.voice_clone_ref_text = self.transcribe_audio(self.voice_clone_ref_audio) or "" if not self.voice_clone_ref_text: print("[WARNING] No reference text available; using x-vector-only clone mode (lower quality).") print(' Pass --transcription "..." for higher-quality in-context cloning.') else: print(f"[OK] Reference text:\n{self.voice_clone_ref_text}") def _init_client(self, url: str, clone: bool = False) -> None: """Initialize a Gradio client and store its API metadata. gradio_client prints its usage info directly to stdout while the client is created and its API metadata loaded, so stdout is swapped for a buffer for the whole process; the captured text is re-emitted at DEBUG level for troubleshooting. """ from gradio_client import Client logger.info("Connecting to Qwen API at %s...", url) old_stdout = sys.stdout captured = io.StringIO() sys.stdout = captured try: try: client = Client(url, httpx_kwargs={"timeout": config.API_TIMEOUT}) except TypeError: # Older gradio_client versions don't support httpx_kwargs. client = Client(url) if clone: self.clone_client = client self.clone_api_info = self._load_api_info(client) else: self.client = client self.api_info = self._load_api_info(client) finally: sys.stdout = old_stdout usage_info = captured.getvalue().strip() if usage_info: logger.debug("Gradio client output for %s:\n%s", url, usage_info) logger.info("Connected to Qwen API") @staticmethod def _load_api_info(client) -> Dict[str, Any]: """Load available API metadata from the Gradio app.""" try: return client.view_api(return_format="dict") except Exception as exc: logger.warning("Unable to read API metadata: %s", exc) return {} def _resolve_api_name(self, *candidates: str, api_info: Optional[Dict[str, Any]] = None) -> str: """Return the first available api_name from candidate list.""" info = api_info if api_info is not None else self.api_info named_endpoints = info.get("named_endpoints", {}) for candidate in candidates: if candidate in named_endpoints: return candidate return candidates[0] def _endpoint_accepts_param(self, api_name: str, param_name: str, api_info: Optional[Dict[str, Any]] = None) -> bool: """Check whether endpoint input schema includes the given parameter.""" info = api_info if api_info is not None else self.api_info endpoint = info.get("named_endpoints", {}).get(api_name, {}) parameters = endpoint.get("parameters", []) return any(parameter.get("parameter_name") == param_name for parameter in parameters) # ------------------------------------------------------------------ # Reference audio transcription (voice clone) # ------------------------------------------------------------------ def transcribe_audio(self, audio_path: str) -> Optional[str]: """Transcribe reference audio locally using an optional Whisper backend.""" return transcribe_reference_audio(audio_path) # ------------------------------------------------------------------ # Chunk generation # ------------------------------------------------------------------ def generate_chunk(self, text: str, chunk_num: int) -> Optional[str]: """Generate one audio chunk; returns its path in the chunks folder. The text is split into sub-requests of at most ``config.CHUNK_SIZE`` words each (the book-level chunker normally guarantees this already; the split is defense in depth against pathological input such as a punctuation-free run of text), and the audio files returned for the sub-requests are concatenated into one chunk file. """ try: sub_texts = split_into_chunks(text, max_words=config.CHUNK_SIZE) if not sub_texts: raise RuntimeError("No text to synthesize") output_path: Optional[Path] = None with tempfile.TemporaryDirectory(prefix="tts_parts_") as parts_dir, \ self._chunk_heartbeat(chunk_num): part_paths = [ self._generate_sub_request(sub_text, parts_dir, sub_num, len(sub_texts), chunk_num) for sub_num, sub_text in enumerate(sub_texts, 1) ] if len(part_paths) == 1: suffix = part_paths[0].suffix or ".wav" output_path = self._chunk_path(chunk_num, suffix) shutil.copy2(part_paths[0], output_path) else: output_path = self._chunk_path(chunk_num, ".wav") concat_audio_files(part_paths, output_path) logger.debug("Chunk %d generated successfully (%d sub-request(s))", chunk_num, len(sub_texts)) return str(output_path) except Exception as exc: logger.error("Qwen chunk processing failed for chunk %d: %s", chunk_num, exc) return None def _generate_sub_request(self, text: str, parts_dir: str, sub_num: int, sub_total: int, chunk_num: int) -> Path: """Run one API generation for ``text``; returns the downloaded audio.""" if sub_total > 1: logger.info("Chunk %d: oversized input split into %d requests " "(sub-request %d/%d)", chunk_num, sub_total, sub_num, sub_total) if self.voice_mode == VOICE_MODE_CUSTOM: result = self._generate_custom_voice(text) elif self.voice_mode == VOICE_MODE_CLONE: result = self._generate_voice_clone(text) else: raise ValueError(f"Unknown voice mode: {self.voice_mode}") if not isinstance(result, (tuple, list)) or not result: raise RuntimeError("Qwen API returned an invalid result") audio_path = result[0] # First element is the audio file path if not isinstance(audio_path, (str, Path)) or not audio_path: raise RuntimeError("Qwen API did not return an audio file path") source = Path(audio_path) if not source.exists(): raise RuntimeError(f"Generated audio file not found: {audio_path}") destination = Path(parts_dir) / f"part_{sub_num:02d}{source.suffix or '.wav'}" shutil.copy2(source, destination) return destination # ------------------------------------------------------------------ # API payloads # ------------------------------------------------------------------ def _generate_custom_voice(self, text: str) -> Tuple: """Generate audio using CustomVoice mode.""" custom_api = self._resolve_api_name("/run_instruct", "/run_custom_voice", "/generate_custom_voice") if custom_api == "/run_instruct": payload = dict( text=text, lang_disp=self.language, spk_disp=speaker_display_name(), instruct=config.INSTRUCT, ) else: payload = dict( text=text, language=self.language, speaker=config.SPEAKER, instruct=config.INSTRUCT, ) if self._endpoint_accepts_param(custom_api, "model_id_cv"): payload["model_id_cv"] = CUSTOM_VOICE_MODEL_ID elif self._endpoint_accepts_param(custom_api, "model_size"): payload["model_size"] = MODEL_SIZE if self._endpoint_accepts_param(custom_api, "seed"): payload["seed"] = self._seed return self.client.predict(**payload, api_name=custom_api) def _ref_audio_payload(self) -> Dict[str, Any]: """Gradio file payload for the reference audio (built once, reused).""" if self._ref_audio_filedata is None: from gradio_client import handle_file self._ref_audio_filedata = handle_file(self.voice_clone_ref_audio) return self._ref_audio_filedata def _generate_voice_clone(self, text: str) -> Tuple: """Generate audio using Voice Clone mode.""" if not Path(self.voice_clone_ref_audio).exists(): raise FileNotFoundError(f"Reference audio not found: {self.voice_clone_ref_audio}") if self.clone_client is None: raise RuntimeError("Voice Clone client is not initialized. Is the Base-model demo running?") clone_api = self._resolve_api_name("/run_voice_clone", "/generate_voice_clone", api_info=self.clone_api_info) use_xvector = config.XVECTOR_ONLY or not self.voice_clone_ref_text if clone_api == "/run_voice_clone": payload = dict( ref_aud=self._ref_audio_payload(), ref_txt=self.voice_clone_ref_text, use_xvec=use_xvector, text=text, lang_disp=self.language, ) else: payload = dict( ref_audio=self._ref_audio_payload(), ref_text=self.voice_clone_ref_text, target_text=text, language=self.language, use_xvector_only=use_xvector, ) optional_params = { "model_size": MODEL_SIZE, "seed": self._seed, } for name, value in optional_params.items(): if self._endpoint_accepts_param(clone_api, name, api_info=self.clone_api_info): payload[name] = value return self.clone_client.predict(**payload, api_name=clone_api) class FasterTTSClient(_BaseTTSClient): """Generates audio chunks through a faster-qwen3-tts server. Talks to the OpenAI-compatible server shipped in the faster-qwen3-tts repository (examples/openai_server.py). The reference voice (ref audio, ref text) and language are configured on the server itself via --ref-audio/--ref-text or a --voices JSON file; this client only sends text. Unlike the Qwen demo, the server performs one generation per request, so long chunks are sub-chunked client-side. """ def __init__(self, voice: Optional[str] = None, api_url: Optional[str] = None): self.voice = voice or config.FASTER_VOICE self.api_url = (api_url or config.FASTER_API_URL).rstrip("/") self._check_health() def _check_health(self) -> None: """Verify the server is reachable and its model is loaded.""" url = f"{self.api_url}/health" try: with urllib.request.urlopen(url, timeout=10) as response: payload = json.loads(response.read().decode("utf-8")) except Exception as exc: raise RuntimeError( f"Faster TTS server not reachable at {url}: {exc}. " "Start the faster-qwen3-tts OpenAI-compatible server first " "(see the 'Faster backend' section of the README)." ) from exc if not payload.get("model_loaded"): raise RuntimeError( "The faster TTS server is running but its model is not loaded yet; " "wait for model download and startup to finish, then retry." ) print(f"[OK] Connected to faster TTS API at {self.api_url} (voice '{self.voice}')") print(f"[INFO] The server silently falls back to its first configured voice if " f"'{self.voice}' is not defined in its voice config (see README).") # ------------------------------------------------------------------ # HTTP requests # ------------------------------------------------------------------ def _request_pcm(self, text: str) -> bytes: """POST one sub-chunk and return raw 16-bit mono PCM bytes.""" url = f"{self.api_url}/v1/audio/speech" payload = json.dumps({ "model": "tts-1", "input": text, "voice": self.voice, "response_format": "pcm", }).encode("utf-8") request = urllib.request.Request( url, data=payload, headers={"Content-Type": "application/json"}, method="POST") try: with urllib.request.urlopen(request, timeout=config.API_TIMEOUT) as response: pcm = response.read() except urllib.error.HTTPError as exc: detail = "" try: detail = exc.read().decode("utf-8", errors="replace")[:200] except Exception: pass raise RuntimeError(f"Faster TTS server returned HTTP {exc.code}: {detail}") from exc except urllib.error.URLError as exc: raise RuntimeError(f"Faster TTS request failed: {exc.reason}") from exc if not pcm: raise RuntimeError("Faster TTS server returned empty audio") return pcm def _request_pcm_with_retry(self, text: str, chunk_num: int, sub_num: int, sub_total: int) -> bytes: """Request one sub-chunk, retrying transient failures.""" for attempt in range(config.MAX_RETRIES): try: return self._request_pcm(text) except Exception as exc: logger.warning("Chunk %d sub-chunk %d/%d attempt %d failed: %s", chunk_num, sub_num, sub_total, attempt + 1, exc) if attempt < config.MAX_RETRIES - 1: time.sleep(2 + 2 * attempt) raise RuntimeError(f"Sub-chunk {sub_num}/{sub_total} failed after " f"{config.MAX_RETRIES} attempts") # ------------------------------------------------------------------ # Chunk generation # ------------------------------------------------------------------ def generate_chunk(self, text: str, chunk_num: int) -> Optional[str]: """Generate one audio chunk; returns its path in the chunks folder.""" try: sub_chunks = split_into_chunks(text, max_words=config.CHUNK_SIZE) if not sub_chunks: raise RuntimeError("No text to synthesize") pcm_parts: List[bytes] = [] with self._chunk_heartbeat(chunk_num): for sub_num, sub_text in enumerate(sub_chunks, 1): pcm = self._request_pcm_with_retry( sub_text, chunk_num, sub_num, len(sub_chunks)) pcm_parts.append(pcm) output_path = self._chunk_path(chunk_num, ".wav") with wave.open(str(output_path), "wb") as wav_file: wav_file.setnchannels(1) wav_file.setsampwidth(2) wav_file.setframerate(SAMPLE_RATE) wav_file.writeframes(b"".join(pcm_parts)) logger.debug("Chunk %d generated (%d sub-chunks)", chunk_num, len(sub_chunks)) return str(output_path) except Exception as exc: logger.error("Faster chunk processing failed for chunk %d: %s", chunk_num, exc) return None class AudioCppTTSClient(_BaseTTSClient): """Generates audio chunks through an audio.cpp audiocpp_server. Talks to the OpenAI-style HTTP API of audiocpp_server, which hosts TTS model families through a native ggml runtime (GGUF weights, no Python serving stack). The server API is family-agnostic; the family and task of the configured model entry are read from GET /v1/models at startup and adapt the request payload (language field style, style instructions) through AUDIOCPP_FAMILY_PROFILES. Three voice modes, all resolved server-side from the request's "voice"/"instructions" fields: - Speaker mode (no ``voice``): Qwen3-TTS only. A built-in CustomVoice speaker name (e.g. "Vivian") is passed through, plus the INSTRUCT style prompt. The server must be configured with the CustomVoice model for this. Families without built-in speakers reject this mode with a hint to pass --voice (or --instructions, see below). - Preset mode (``voice=NAME``): a voice configured on the server (``voice_presets`` or ``voice_dir`` in its config, e.g. a cloning reference). The name is validated against GET /v1/audio/voices at startup because an unresolvable name would silently fall back to plain TTS on a clone-based model instead of failing. When AUDIOCPP_CLONE_MODEL_ID names a second server entry of the same family (typically the Qwen Base model), preset requests are routed to it. Only the entry actually used needs to exist on the server: a clone-only (Base) server works for --voice runs, while speaker mode on such a server fails with a hint to pass --voice. - Voice design (task "vdes" entries, e.g. Qwen3-TTS VoiceDesign): the voice is described in natural language through ``instructions``, which is required and sent with every request (no ``voice`` field). A constant per-run seed keeps the designed voice consistent across chunk boundaries. ``instructions`` also works on non-design entries, where it acts as a generic style/delivery instruction (voice control): families that read it (OmniVoice, Qwen3-TTS CustomVoice, ...) shape the voice or delivery accordingly, and others ignore it. On instruction-conditioned families without built-in speakers it may replace --voice entirely (the instruction defines the voice). Extra request options (``--option KEY=VALUE``, e.g. emotion, voice_id, speed) are forwarded verbatim in the request's "options" object, which is the server's generic pass-through for per-model controls. Chunking: the server does its own long-form text chunking for every family (its ``text_chunk_size`` option, with a per-family default), so by default each chapter is sent as a single request and the audio comes back already stitched. With ``chunk_text=True`` (the --chunk CLI flag), text is instead split client-side into CHUNK_SIZE-word sub-requests, which may needlessly double-chunk — the warning is printed by the CLI. Each response is a complete WAV file, so sub-request audio is concatenated with the same lossless path used for the Qwen client. """ def __init__(self, voice: Optional[str] = None, language: Optional[str] = None, api_url: Optional[str] = None, chunk_text: bool = False, model_id: Optional[str] = None, instructions: Optional[str] = None, request_options: Optional[Dict[str, str]] = None): self.api_url = (api_url or config.AUDIOCPP_API_URL).rstrip("/") # Per-run model selection: the --model CLI flag overrides config; an # empty value is resolved at connect time when the server hosts exactly # one entry, so multi-model servers don't require editing config.py. self.model_id = (model_id if model_id is not None else config.AUDIOCPP_MODEL_ID) or "" self._model_id_explicit = bool(self.model_id) # Validate before connecting so bad values fail fast without a server. self.language = normalize_language( language if language is not None else config.LANGUAGE) # One seed value per run, reused for every request (see # _resolve_request_seed). Unlike the Qwen demo, audio.cpp has no # negative "randomize" seed, so a negative value means "send no seed # at all" (see _request_wav) and the server randomizes. self._seed = _resolve_request_seed() self.preset_mode = bool(voice) self.voice = voice or speaker_display_name() # Style/voice-design instruction sent with every request (the CLI # --instructions flag overrides AUDIOCPP_INSTRUCTIONS in config.py). # For task "vdes" entries it describes the voice to design; for other # families it is a generic style instruction when the model reads one. self.instructions = (instructions if instructions is not None else config.AUDIOCPP_INSTRUCTIONS or "").strip() # Free-form per-request options (--option KEY=VALUE) forwarded in the # request's "options" object; models ignore keys they don't know. self.request_options: Dict[str, str] = dict(request_options or {}) # Both set during _connect once the entry's task is known: design_mode # for "vdes" entries, instruction_voice when a family without built-in # speakers gets its voice from the instruction alone (no voice field). self.design_mode = False self.instruction_voice = False # When False (default), each chapter is sent as one request and the # server does its own long-form chunking (text_chunk_size); when True, # text is split client-side into CHUNK_SIZE-word sub-requests first. self.chunk_text = bool(chunk_text) # Family and task of the selected model entry and the family's request # profile; all are resolved from GET /v1/models during _connect. self.family = "" self.task = AUDIOCPP_TASK_TTS self.profile = AUDIOCPP_DEFAULT_FAMILY_PROFILE self._connect() # ------------------------------------------------------------------ # Connection # ------------------------------------------------------------------ def _connect(self) -> None: """Health-check the server and resolve the model, family, task, and voice. Speaker mode is only offered to families with built-in speakers (Qwen3-TTS); every other family must select a server-side voice with --voice or describe one with --instructions, so it fails fast with a hint instead of silently synthesizing with a random default voice. Voice design entries (task "vdes") require --instructions and reject --voice. """ self._check_health() models = self._list_models() self._auto_pick_model_id(models) if self.preset_mode: self._select_model(models) self._require_model_id(models) self._resolve_family(models) self._resolve_task(models) if self.task not in AUDIOCPP_SYNTHESIS_TASKS: available = ", ".join(model["id"] for model in models) or "none" raise RuntimeError( f"The audio.cpp model '{self.model_id}' has task " f"'{self.task}'; audiobook.py can only synthesize with TTS " f"model entries (tasks {', '.join(AUDIOCPP_SYNTHESIS_TASKS)}). " f"Pick a synthesis entry with --model (available: {available})." ) if self.design_mode: if self.preset_mode: raise RuntimeError( f"--voice cannot be used with the voice design model " f"'{self.model_id}': the voice is described by the " "--instructions text instead (see README).") if not self.instructions: raise RuntimeError( f"The audio.cpp model '{self.model_id}' (family " f"'{self.family}') is a voice design model: pass a " "description of the voice to synthesize with, e.g. " '--instructions "A warm adult female narrator with a ' 'British accent" (see README).') print(f"[OK] Connected to audio.cpp server at {self.api_url} " f"(model '{self.model_id}', family '{self.family}', " "voice design)") print(f"[INFO] Designing the voice from: {self.instructions}") elif self.preset_mode: self._check_voice() print(f"[OK] Connected to audio.cpp server at {self.api_url} " f"(model '{self.model_id}', family '{self.family}', " f"voice '{self.voice}')") elif self.profile.builtin_speakers: print(f"[OK] Connected to audio.cpp server at {self.api_url} " f"(model '{self.model_id}', family '{self.family}', " f"speaker '{self.voice}')") print("[INFO] Speaker mode expects the server to be configured with the " "CustomVoice model; with the Base model the speaker name is ignored " "and a random default voice is used (see README).") elif self.instructions: # Families without built-in speakers can still get their voice # from the instruction alone (e.g. OmniVoice voice design). self.instruction_voice = True print(f"[OK] Connected to audio.cpp server at {self.api_url} " f"(model '{self.model_id}', family '{self.family}', " "instruction voice)") print(f"[INFO] Designing the voice from: {self.instructions}") else: raise RuntimeError( f"The audio.cpp model '{self.model_id}' (family " f"'{self.family}') has no built-in speakers, so its voice " "must come from the server: rerun with --voice NAME " "matching a voice_preset or voice_dir entry in the server " "config, or describe a voice with --instructions for " "families that support it (see README).") if self.instructions and not self.design_mode and not self.instruction_voice: print(f"[INFO] Sending instruction with every request: {self.instructions}") print("[INFO] Its effect (style, emotion, delivery) depends on the " "model family; models without instruction support ignore it.") def _get_json(self, path: str, timeout: int = 10) -> Dict[str, Any]: """GET a JSON document from the server.""" url = f"{self.api_url}{path}" try: with urllib.request.urlopen(url, timeout=timeout) as response: return json.loads(response.read().decode("utf-8")) except urllib.error.HTTPError as exc: detail = "" try: detail = exc.read().decode("utf-8", errors="replace")[:200] except Exception: pass raise RuntimeError( f"audio.cpp server returned HTTP {exc.code} for {path}: {detail}") from exc except urllib.error.URLError as exc: raise RuntimeError(f"audio.cpp request failed for {path}: {exc.reason}") from exc def _check_health(self) -> None: """Verify the server is reachable and reports healthy.""" try: payload = self._get_json("/health") except Exception as exc: raise RuntimeError( f"audio.cpp server not reachable at {self.api_url}: {exc}. " "Start audiocpp_server first (see the 'audio.cpp backend' " "section of the README)." ) from exc if payload.get("status") != "ok": raise RuntimeError( f"The audio.cpp server at {self.api_url} reports status " f"{payload.get('status')!r} instead of 'ok'") def _list_models(self) -> List[Dict[str, str]]: """Fetch the (id, family, task) triples reported by the server.""" try: payload = self._get_json("/v1/models") except Exception as exc: raise RuntimeError( f"The audio.cpp server at {self.api_url} did not answer " f"/v1/models: {exc}") from exc entries = payload.get("data") or [] models: List[Dict[str, str]] = [] for entry in entries: if isinstance(entry, dict) and entry.get("id"): models.append({ "id": entry["id"], "family": entry.get("family") or "", "task": entry.get("task") or "", }) return models def _auto_pick_model_id(self, models: List[Dict[str, str]]) -> None: """Resolve an empty model id when the server hosts exactly one entry. Multi-model servers generated with several lazily-loaded entries can be used without editing converter/config.py: leave AUDIOCPP_MODEL_ID (and ``--model``) unset, and the single hosted entry is chosen automatically. With more than one entry an explicit choice is required (via ``--model`` or AUDIOCPP_MODEL_ID), since guessing would risk synthesizing a whole book with the wrong family. """ if self.model_id: return if len(models) == 1: self.model_id = models[0]["id"] logger.info( "AUDIOCPP_MODEL_ID is unset; using the only server entry '%s'", self.model_id) else: logger.debug( "AUDIOCPP_MODEL_ID is unset and the server hosts %d entries; " "an explicit --model or config id is required", len(models)) def _require_model_id(self, models: List[Dict[str, str]]) -> None: """Verify the model id chosen for this run exists on the server. Speaker mode needs AUDIOCPP_MODEL_ID (the CustomVoice entry). Preset mode validates whichever id _select_model resolved, so a server hosting only a cloning model works for --voice. """ model_ids = [model["id"] for model in models] if self.model_id and self.model_id in model_ids: return configured = ", ".join(model_ids) or "none" if not self.model_id: raise RuntimeError( f"The audio.cpp server at {self.api_url} hosts {len(model_ids)} " f"model entries ({configured}); audiobook.py needs to know which " "one to use. Pass --model when converting, or set " "AUDIOCPP_MODEL_ID in converter/config.py to one of them " "(see README)." ) if self.preset_mode: raise RuntimeError( f"The audio.cpp server at {self.api_url} has no model id " f"'{self.model_id}' or clone model id " f"'{config.AUDIOCPP_CLONE_MODEL_ID}' (configured: {configured}). " "Add a TTS model entry for the family you want to the server " "config and match AUDIOCPP_MODEL_ID / AUDIOCPP_CLONE_MODEL_ID " "in converter/config.py to its id, or select it per run with " "--model (see README)." ) raise RuntimeError( f"The audio.cpp server at {self.api_url} has no model id " f"'{self.model_id}' (configured: {configured}). Speaker mode needs " "the Qwen3-TTS CustomVoice model: add a qwen3_tts model entry to " "the server config and match AUDIOCPP_MODEL_ID in converter/config.py to its " "id (or pass --model), or rerun with --voice to use a voice preset " "on any TTS model (see README)." ) def _select_model(self, models: List[Dict[str, str]]) -> None: """Pick the model for preset (cloning) requests. Defaults to the primary model id. When AUDIOCPP_CLONE_MODEL_ID is configured and present on the server, preset requests are routed to it instead, so one server can host the CustomVoice model for speaker mode and the Base model for cloning (Qwen3-TTS setups). A clone id that names a model of a different family is ignored with a warning, since preset requests must synthesize with the family the run is configured for. """ clone_model_id = config.AUDIOCPP_CLONE_MODEL_ID if not clone_model_id or clone_model_id == self.model_id: return families = {model["id"]: model["family"] for model in models} if clone_model_id not in families: # A qwen3_tts primary without its clone entry silently degrades # (presets are ignored on the CustomVoice model), so that case # keeps the warning; single-model servers of other families are # the normal configuration and only get a debug note. primary_is_qwen = (families.get(self.model_id) or AUDIOCPP_FAMILY_QWEN3_TTS) \ == AUDIOCPP_FAMILY_QWEN3_TTS if primary_is_qwen: logger.warning( "AUDIOCPP_CLONE_MODEL_ID %r is not configured on the audio.cpp " "server; preset requests use '%s' instead", clone_model_id, self.model_id) else: logger.debug( "AUDIOCPP_CLONE_MODEL_ID %r is not configured on the audio.cpp " "server; preset requests use '%s' instead", clone_model_id, self.model_id) return primary_family = families.get(self.model_id) clone_family = families[clone_model_id] if primary_family and clone_family and primary_family != clone_family: logger.warning( "AUDIOCPP_CLONE_MODEL_ID %r hosts family %r, but " "AUDIOCPP_MODEL_ID %r hosts %r; preset requests stay on " "'%s'. Point both ids at the same model entry in " "converter/config.py (single-model servers use the same id " "for both)", clone_model_id, clone_family, self.model_id, primary_family, self.model_id) return self.model_id = clone_model_id def _resolve_family(self, models: List[Dict[str, str]]) -> None: """Resolve the selected model's family and its request profile. The family comes from GET /v1/models. Servers that predate the family field served Qwen3-TTS only, so a missing family is treated as qwen3_tts, which also preserves this client's legacy behavior against those versions. """ entry = next( (model for model in models if model["id"] == self.model_id), None) family = (entry["family"] if entry is not None else "") or "" if not family: family = AUDIOCPP_FAMILY_QWEN3_TTS logger.debug("Model '%s' reported no family; assuming qwen3_tts", self.model_id) self.family = family self.profile = AUDIOCPP_FAMILY_PROFILES.get( family, AUDIOCPP_DEFAULT_FAMILY_PROFILE) if family not in AUDIOCPP_FAMILY_PROFILES: logger.info( "audio.cpp family '%s' has no dedicated profile; using the " "generic profile (voice cloning via --voice, model-detected " "language)", family) def _resolve_task(self, models: List[Dict[str, str]]) -> None: """Resolve the selected model's task (tts, clon, vdes, ...) and set design mode for voice design entries. The task comes from GET /v1/models and is fixed per server entry by its server.json config (a VoiceDesign model must be hosted with "task": "vdes"). Servers that predate the task field hosted plain TTS models, so a missing task is treated as tts. """ entry = next( (model for model in models if model["id"] == self.model_id), None) task = (entry["task"] if entry is not None else "") or "" if not task: task = AUDIOCPP_TASK_TTS logger.debug("Model '%s' reported no task; assuming tts", self.model_id) self.task = task self.design_mode = task == AUDIOCPP_TASK_VDES def _check_voice(self) -> None: """Verify the requested voice is available on the server. A voice name that matches no server preset or voice-library wav would be passed through to the model as a cached voice id; on the Base (cloning) model that is silently ignored and plain TTS audio comes back, so preset names are validated up front. When the voices endpoint cannot be queried, validation is skipped with a warning rather than blocking the run. """ query = urllib.parse.urlencode({"model": self.model_id}) try: payload = self._get_json(f"/v1/audio/voices?{query}") except Exception as exc: logger.warning("Could not list server voices; skipping voice " "validation: %s", exc) return voices = payload.get("voices") or [] if self.voice not in voices: available = ", ".join(str(v) for v in voices) or "none" raise RuntimeError( f"Voice '{self.voice}' is not available on the audio.cpp server " f"(available: {available}). Configure it as a voice_preset or " "voice_dir entry in the server config, or pass a listed name " "with --voice (see README)." ) # ------------------------------------------------------------------ # HTTP requests # ------------------------------------------------------------------ def _request_wav(self, text: str) -> bytes: """POST one sub-chunk and return the raw WAV bytes. The request timeout scales with the text length when a whole chapter is sent in one request (no client-side chunking), since a long chapter means many minutes of audio generated in one go. """ url = f"{self.api_url}/v1/audio/speech" payload: Dict[str, Any] = { "model": self.model_id, "input": text, } # Design models take no voice field (the voice comes from the # instruction); instruction-voice runs on families without built-in # speakers omit it too, since no speaker or preset was requested. if not self.design_mode and not self.instruction_voice: payload["voice"] = self.voice if self.profile.language_style == AUDIOCPP_LANG_DISPLAY: payload["language"] = self.language elif self.profile.language_style == AUDIOCPP_LANG_ISO: iso_code = LANGUAGE_ISO_CODES.get(self.language) if iso_code: payload["language"] = iso_code else: # "Auto": no code to send, so let the server pick its default. logger.debug("%s: no language code for %r; omitted from request", self.family, self.language) if self._seed >= 0: # audio.cpp has no negative "randomize" seed; a negative seed # means "let the server randomize", so the field is omitted. payload["seed"] = self._seed if self.instructions: # Explicit voice-design or style instruction (required for task # "vdes" entries; a Ctrl/style control on families that read it). payload["instructions"] = self.instructions elif not self.preset_mode and config.INSTRUCT \ and self.profile.sends_instructions: # Style instruction for the Qwen3-TTS CustomVoice speakers; # ignored by the Base (cloning) model and other families. payload["instructions"] = config.INSTRUCT if self.request_options: # Generic per-model controls (--option KEY=VALUE): forwarded # verbatim; the model ignores keys it does not know. payload["options"] = dict(self.request_options) request = urllib.request.Request( url, data=json.dumps(payload).encode("utf-8"), headers={"Content-Type": "application/json"}, method="POST") timeout = config.API_TIMEOUT if not self.chunk_text: # Estimated audio duration at 150 wpm, doubled plus a minute of # slack, bounded below by the configured per-request timeout. estimated_seconds = 60.0 * len(text.split()) / _ESTIMATED_WORDS_PER_MINUTE timeout = max(timeout, int(estimated_seconds * 2) + 60) try: with urllib.request.urlopen(request, timeout=timeout) as response: wav = response.read() except urllib.error.HTTPError as exc: detail = "" try: detail = exc.read().decode("utf-8", errors="replace")[:200] except Exception: pass raise RuntimeError(f"audio.cpp server returned HTTP {exc.code}: {detail}") from exc except urllib.error.URLError as exc: raise RuntimeError(f"audio.cpp request failed: {exc.reason}") from exc if len(wav) < 12 or wav[:4] != b"RIFF" or wav[8:12] != b"WAVE": raise RuntimeError("audio.cpp server returned audio that is not a WAV file") return wav def _request_wav_with_retry(self, text: str, chunk_num: int, sub_num: int, sub_total: int) -> bytes: """Request one sub-chunk, retrying transient failures.""" for attempt in range(config.MAX_RETRIES): try: return self._request_wav(text) except Exception as exc: logger.warning("Chunk %d sub-chunk %d/%d attempt %d failed: %s", chunk_num, sub_num, sub_total, attempt + 1, exc) if attempt < config.MAX_RETRIES - 1: time.sleep(2 + 2 * attempt) raise RuntimeError(f"Sub-chunk {sub_num}/{sub_total} failed after " f"{config.MAX_RETRIES} attempts") # ------------------------------------------------------------------ # Chunk generation # ------------------------------------------------------------------ def generate_chunk(self, text: str, chunk_num: int) -> Optional[str]: """Generate one audio chunk; returns its path in the chunks folder. By default the whole text goes out as a single request and the server does its own long-form chunking (see the class docstring). With ``chunk_text=True`` (--chunk), the text is split into sub-requests of at most ``config.CHUNK_SIZE`` words each; each sub-request returns a complete WAV file and the parts are concatenated into one chunk file. """ try: if self.chunk_text: sub_texts = split_into_chunks(text, max_words=config.CHUNK_SIZE) elif text.strip(): sub_texts = [text] else: sub_texts = [] if not sub_texts: raise RuntimeError("No text to synthesize") output_path: Optional[Path] = None with tempfile.TemporaryDirectory(prefix="tts_parts_") as parts_dir, \ self._chunk_heartbeat( chunk_num, label=None if self.chunk_text else "Request"): part_paths = [] for sub_num, sub_text in enumerate(sub_texts, 1): wav = self._request_wav_with_retry( sub_text, chunk_num, sub_num, len(sub_texts)) destination = Path(parts_dir) / f"part_{sub_num:02d}.wav" destination.write_bytes(wav) part_paths.append(destination) if len(part_paths) == 1: output_path = self._chunk_path(chunk_num, ".wav") shutil.copy2(part_paths[0], output_path) else: output_path = self._chunk_path(chunk_num, ".wav") concat_audio_files(part_paths, output_path) logger.debug("Chunk %d generated successfully (%d sub-request(s))", chunk_num, len(sub_texts)) return str(output_path) except Exception as exc: logger.error("audio.cpp chunk processing failed for chunk %d: %s", chunk_num, exc) return None