"""Client for the audio.cpp audiocpp_server (native ggml TTS families).""" import json import logging import shutil import tempfile import urllib.error import urllib.parse import urllib.request from pathlib import Path from typing import Any, Dict, List, Optional from .. import config from ..audio import concat_audio_files from ..chunking import split_into_chunks from .base import (BaseTTSClient, ConversionCancelled, NonRetryableTTSError, resolve_request_seed) from .languages import LANGUAGE_ISO_CODES, normalize_language from .speakers import is_builtin_speaker, speaker_display_name_for logger = logging.getLogger(__name__) # How the "language" request field is expressed by a family. AUDIOCPP_LANG_DISPLAY = "display" # Qwen display names, e.g. "English" AUDIOCPP_LANG_ISO = "iso" # ISO 639-1 codes, e.g. "en" AUDIOCPP_LANG_OMIT = "omit" # no language field; the model detects it # The Qwen3-TTS family. Unlike every other family (one model type each), # qwen3_tts hosts several model *types* under one family id, distinguished # only by the server entry's id/task: the CustomVoice model (built-in # speakers, e.g. Vivian/Ryan), the Base model (voice cloning via a # server-side preset), and the VoiceDesign model (task "vdes"). The # per-entry voice capability below (audiocpp_entry_voice_capability) # resolves which is which, driving both the Convert form (which voice # list to show) and the converter's mode selection. AUDIOCPP_FAMILY_QWEN3_TTS = "qwen3_tts" # Server model entry tasks this client can synthesize audiobooks with, # taken from GET /v1/models (the "task" field of each entry; a missing task # is treated as "tts" — a harmless generic default). "vdes" entries are # voice design models: the voice is described with --instructions instead # of coming from a speaker or a reference clip. Entries with any other task # (asr, vc, diar, ...) are rejected at connect time with a hint to pick a # synthesis entry. AUDIOCPP_TASK_TTS = "tts" AUDIOCPP_TASK_VDES = "vdes" AUDIOCPP_SYNTHESIS_TASKS = (AUDIOCPP_TASK_TTS, "clon", AUDIOCPP_TASK_VDES) # The voice capability of a server model entry — how its voice is supplied. # Resolved per entry from (family, task, id) by # audiocpp_entry_voice_capability; drives both the Convert form (which # voice list to show) and the converter (speaker vs preset vs design mode). # Most families are clone-only; only the Qwen3-TTS CustomVoice entry has # built-in speakers, and only VoiceDesign entries take a description. AUDIOCPP_VOICE_SPEAKER = "speaker" # built-in speaker name (Qwen CustomVoice) AUDIOCPP_VOICE_CLONE = "clone" # server-side preset / voice_dir (Base, others) AUDIOCPP_VOICE_DESIGN = "design" # voice described by --instructions (vdes) # HTTP error body fragments identifying deterministic request-configuration # problems: the identical request will fail on every retry, so the chunk # loop must give up immediately instead of burning its attempt budget. # Matched case-insensitively against the server's error message; the # fragments come from audio.cpp itself, so they hold for every hosted # family (none are model-specific). AUDIOCPP_NON_RETRYABLE_ERRORS = ( # Cloning without the reference transcript (Qwen3-TTS Base ICL mode): # the server-side voice has reference audio but no transcript for it. "requires reference text", # The server cannot resolve a model contract for the family (its own # hint text about model_specs/--model-spec-override follows the fragment). "model contract spec not found for family", "does not embed an audio.cpp model spec", "embeds a legacy model spec", # The request named a model the server does not host. "unknown model id", ) _REFERENCE_TEXT_FRAGMENT = AUDIOCPP_NON_RETRYABLE_ERRORS[0] def _server_error_message(detail: str) -> str: """The server's error message from an HTTP error body, else the body. The speech endpoint wraps failures as {"error": {"message": ...}}; the inner message is what matches AUDIOCPP_NON_RETRYABLE_ERRORS and what the user should see. Unparseable bodies are returned as-is. """ try: payload = json.loads(detail) except ValueError: return detail if isinstance(payload, dict): error = payload.get("error") if isinstance(error, dict) and isinstance(error.get("message"), str): return error["message"] if isinstance(error, str): return error return detail def _reference_text_error(voice: Optional[str], server_message: str) -> str: """Actionable message for the missing-reference-transcript failure. The server resolved the requested voice to reference audio but has no transcript for it, so its ICL voice-clone path rejects every request. The fix is server-side data, not a client retry: prompt_text (or the voice preset's reference_text) supplies it, read per request, so no server restart is needed. x_vector_only_mode is the transcript-free escape hatch, at the cost of speaker similarity. """ name = f"'{voice}'" if voice else "the requested voice" return ( f"The audio.cpp server cannot clone voice {name}: its reference " "audio has no transcript, and this model family's voice cloning " f"requires one ({server_message}). Add the transcript to the " "prompt_text file in the server's voice directory (one " "'|' line per voice) or set reference_text on " "the voice preset in server.json; the server reads it per request, " "no restart needed. Re-running the audio.cpp setup re-transcribes " "the reference wavs with whisper. Alternatively rerun with " "--option x_vector_only_mode=true to clone from the speaker " "embedding alone (no transcript needed; lower similarity)." ) def audiocpp_request_error(status: int, detail: str, voice: Optional[str] = None) -> Exception: """The exception for a failed audio.cpp speech request. Deterministic request-configuration errors (a fragment in AUDIOCPP_NON_RETRYABLE_ERRORS) become NonRetryableTTSError so the chunk retry loop skips attempts that cannot succeed; everything else returns the plain RuntimeError the retry loop has always retried. """ message = _server_error_message(detail) lowered = message.lower() if _REFERENCE_TEXT_FRAGMENT in lowered: return NonRetryableTTSError( _reference_text_error(voice, message)) if any(fragment in lowered for fragment in AUDIOCPP_NON_RETRYABLE_ERRORS): return NonRetryableTTSError( f"audio.cpp server returned HTTP {status} (not retryable): " f"{message}") return RuntimeError(f"audio.cpp server returned HTTP {status}: {detail}") class AudioCppFamilyProfile: """Request conventions of one audio.cpp model family. Language style and whether the family reads a style/instruction prompt; these are family-level (every entry of a family shares them). Whether a *specific entry* has built-in speakers is an entry-level concern, decided by audiocpp_entry_voice_capability, not this profile. """ def __init__(self, language_style: str = AUDIOCPP_LANG_OMIT, sends_instructions: bool = False): self.language_style = language_style self.sends_instructions = sends_instructions # Generic profile for families not listed in AUDIOCPP_FAMILY_PROFILES: # clone-only, no style instructions, and no language field (the model # detects the language itself). Describes higgs_audio_tts, voxcpm2, # fish_audio, dots_tts, dramabox, omnivoice, outetts, glm_tts, miotts, # moss_tts_*, pocket_tts, vibevoice, ... as well as families added to # audio.cpp after this table was written. AUDIOCPP_DEFAULT_FAMILY_PROFILE = AudioCppFamilyProfile() AUDIOCPP_FAMILY_PROFILES = { AUDIOCPP_FAMILY_QWEN3_TTS: AudioCppFamilyProfile( language_style=AUDIOCPP_LANG_DISPLAY, sends_instructions=True, ), # Families whose language option takes a code (e.g. "en") instead of # a Qwen display name; otherwise clone-only like the default profile. "chatterbox": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO), "confucius4_tts": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO), "index_tts2": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO), "magpie_tts": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO), "supertonic": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO), } def audiocpp_entry_voice_capability(family: str, task: str, model_id: str) -> str: """How a server model entry's voice is supplied — speaker/clone/design. Resolved from the entry's family, task and id — the same {id, family, task} triple GET /v1/models reports, so it works for local server.json entries and remote live-queried entries alike. Qwen3-TTS is the one family hosting several model *types* under one family id: the CustomVoice model (id contains "customvoice") has built-in speakers, the Base model and any other entry are clone-only, and VoiceDesign entries (task "vdes") take a description. Every other family is clone-only. """ if task == AUDIOCPP_TASK_VDES: return AUDIOCPP_VOICE_DESIGN if family == AUDIOCPP_FAMILY_QWEN3_TTS \ and "customvoice" in (model_id or "").lower(): return AUDIOCPP_VOICE_SPEAKER return AUDIOCPP_VOICE_CLONE class AudioCppTTSClient(BaseTTSClient): """Generates audio chunks through an audio.cpp audiocpp_server. Talks to the OpenAI-style HTTP API of audiocpp_server, which hosts TTS model families through a native ggml runtime (GGUF weights, no Python serving stack). The server API is family-agnostic; the family and task of the configured model entry are read from GET /v1/models at startup and adapt the request payload (language field style, style instructions) through AUDIOCPP_FAMILY_PROFILES. The entry's voice capability (audiocpp_entry_voice_capability: speaker / clone / design) decides how its voice is supplied; all three are resolved server-side from the request's "voice"/"instructions" fields: - Speaker mode (--voice with a built-in speaker name): Qwen3-TTS CustomVoice only. A built-in speaker name (e.g. "Vivian") is passed through. The selected entry must be the CustomVoice model (capability == speaker); a speaker name on a non-speaker entry is treated as a server-side preset instead. - Preset mode (--voice NAME): a voice configured on the server (``voice_presets`` or ``voice_dir`` in its config, e.g. a cloning reference). The name is validated against GET /v1/audio/voices at startup because an unresolvable name would silently fall back to plain TTS on a clone-based model instead of failing. To clone on a Qwen3-TTS setup, select the Base model entry with --model and pass a preset voice. - Voice design (task "vdes" entries, e.g. Qwen3-TTS VoiceDesign): the voice is described in natural language through ``instructions``, which is required and sent with every request (no ``voice`` field). A constant per-run seed keeps the designed voice consistent across chunk boundaries. The entry's capability decides how an explicit --voice is read: on a speaker-capable entry a name that matches a built-in speaker selects speaker mode, and every other name is a server-side preset. The entry's capability picks the mode: design entries require --instructions; speaker entries require --voice naming a built-in CustomVoice speaker; clone entries (the Base model, and every other family) require --voice with a server-side preset — all fail fast with a hint instead of silently synthesizing with a random default voice. ``instructions`` also works on non-design entries, where it acts as a generic style/delivery instruction (voice control): families that read it (OmniVoice, Qwen3-TTS CustomVoice, ...) shape the voice or delivery accordingly, and others ignore it. On instruction-conditioned families without built-in speakers it may replace --voice entirely (the instruction defines the voice). Extra request options (``--option KEY=VALUE``, e.g. emotion, voice_id, speed) are forwarded verbatim in the request's "options" object, which is the server's generic pass-through for per-model controls. Chunking: text is split client-side into sub-requests of at most config.CHUNK_SIZE words each; each sub-request returns a complete WAV file and the parts are concatenated with the same lossless path used for the Qwen client. """ def __init__(self, chunks_dir: Path, voice: Optional[str] = None, language: Optional[str] = None, api_url: Optional[str] = None, model_id: Optional[str] = None, instructions: Optional[str] = None, request_options: Optional[Dict[str, str]] = None, quiet: bool = False): super().__init__(chunks_dir, quiet=quiet) self.api_url = (api_url or config.AUDIOCPP_API_URL).rstrip("/") # Per-run model selection: the --model CLI flag (or the Generate # form's Model pick). An empty value is resolved at connect time # when the server hosts exactly one entry, so single-model servers # don't require --model. self.model_id = (model_id or "").strip() self._model_id_explicit = bool(self.model_id) # Validate before connecting so bad values fail fast without a server. self.language = normalize_language( language if language is not None else config.LANGUAGE) # One seed value per run, reused for every request (see # resolve_request_seed). Unlike the Qwen demo, audio.cpp has no # negative "randomize" seed, so a negative value means "send no seed # at all" (see _request_wav) and the server randomizes. self._seed = resolve_request_seed() # Voice selection (the --voice name). preset_mode / speaker_mode are # resolved in _connect: a --voice that names a built-in CustomVoice # speaker on a speaker-capable entry selects speaker mode; every # other name (and any name on a clone-capable entry) is a server-side # preset. The request's "voice" field (self.voice) is filled in # _connect per the mode. self.preset_mode = False self.speaker_mode = False self.voice = voice or None # Style/voice-design instruction sent with every request (the CLI # --instructions flag / the Generate form's Instructions field). # For task "vdes" entries it describes the voice to design; for other # families it is a generic style instruction when the model reads one. self.instructions = (instructions or "").strip() # Free-form per-request options (--option KEY=VALUE) forwarded in the # request's "options" object; models ignore keys they don't know. self.request_options: Dict[str, str] = dict(request_options or {}) # Set during _connect: design_mode for "vdes" entries, instruction_voice # when a family without built-in speakers gets its voice from the # instruction alone (no voice field). self.voice is also finalized # there (the speaker/preset name). self.design_mode = False self.instruction_voice = False # Family and task of the selected model entry and the family's request # profile; all are resolved from GET /v1/models during _connect. self.family = "" self.task = AUDIOCPP_TASK_TTS self.profile = AUDIOCPP_DEFAULT_FAMILY_PROFILE self._connect() # ------------------------------------------------------------------ # Connection # ------------------------------------------------------------------ def _connected(self, mode: str) -> None: """Report the resolved connection (MODE: speaker/voice/... label).""" self._report(f"[OK] Connected to audio.cpp server at {self.api_url} " f"(model '{self.model_id}', family '{self.family}', " f"{mode})") def _connect(self) -> None: """Health-check the server and resolve the model, family, task, and voice. The entry's voice capability (audiocpp_entry_voice_capability, from family/task/id) plus the caller's --voice/--instructions pick the mode. An explicit --voice on a speaker-capable (CustomVoice) entry that names a built-in speaker selects speaker mode; every other --voice is a server-side preset, validated against the server's voice library. Without a --voice, design entries require --instructions and every other capability requires --voice — the run fails fast with a hint instead of silently synthesizing with a random default voice. """ self._check_health() models = self._list_models() self._auto_pick_model_id(models) if self.voice is not None: # Explicit --voice: decide between speaker mode and a server-side # preset. A name matching a built-in CustomVoice speaker on a # speaker-capable primary selects speaker mode; every other name # (and any name when the primary entry is absent) is a preset, # validated against the server's voice library. primary = next((m for m in models if m["id"] == self.model_id), None) if primary is not None: self._resolve_family(models) self._resolve_task(models) capability = audiocpp_entry_voice_capability( self.family, self.task, self.model_id) if capability == AUDIOCPP_VOICE_SPEAKER \ and is_builtin_speaker(self.voice): self._require_synthesis_task(models) self.voice = speaker_display_name_for(self.voice) self.speaker_mode = True self._connected(f"speaker '{self.voice}'") if not self.speaker_mode: # Server-side preset (--voice): validate it. self.preset_mode = True self._require_model_id(models) self._resolve_family(models) self._resolve_task(models) self._require_synthesis_task(models) if self.design_mode: raise RuntimeError( f"--voice cannot be used with the voice design model " f"'{self.model_id}': the voice is described by the " "--instructions text instead (see README).") self._check_voice() self._connected(f"voice '{self.voice}'") else: # No flag: the entry's capability picks the default mode. self._require_model_id(models) self._resolve_family(models) self._resolve_task(models) self._require_synthesis_task(models) capability = audiocpp_entry_voice_capability( self.family, self.task, self.model_id) if self.design_mode: if not self.instructions: raise RuntimeError( f"The audio.cpp model '{self.model_id}' (family " f"'{self.family}') is a voice design model: pass a " "description of the voice to synthesize with, e.g. " '--instructions "A warm adult female narrator with a ' 'British accent" (see README).') self._connected("voice design") self._report(f"[INFO] Designing the voice from: {self.instructions}") elif capability == AUDIOCPP_VOICE_SPEAKER: # No --voice on a CustomVoice entry: refuse instead of # guessing a built-in speaker. raise RuntimeError( f"The audio.cpp model '{self.model_id}' (family " f"'{self.family}') serves built-in speakers: pass " "--voice NAME with one of them (e.g. Vivian, Ryan, " "Uncle Fu) to synthesize with it (see README).") elif self.instructions: # Families without built-in speakers can still get their voice # from the instruction alone (e.g. OmniVoice voice design). self.instruction_voice = True self._connected("instruction voice") self._report(f"[INFO] Designing the voice from: {self.instructions}") else: raise RuntimeError( f"The audio.cpp model '{self.model_id}' (family " f"'{self.family}') has no built-in speakers, so its voice " "must come from the server: rerun with --voice NAME " "matching a voice_preset or voice_dir entry in the server " "config, or describe a voice with --instructions for " "families that support it, or select the CustomVoice entry " "for built-in speakers (see README).") if self.instructions and not self.design_mode and not self.instruction_voice: self._report(f"[INFO] Sending instruction with every request: {self.instructions}") self._report("[INFO] Its effect (style, emotion, delivery) depends on the " "model family; models without instruction support ignore it.") if config.AUDIOCPP_UNLOAD_MODELS: self._unload_server_models() def _require_synthesis_task(self, models: List[Dict[str, str]]) -> None: """Reject model entries whose task is not a TTS synthesis task.""" if self.task in AUDIOCPP_SYNTHESIS_TASKS: return available = ", ".join(model["id"] for model in models) or "none" raise RuntimeError( f"The audio.cpp model '{self.model_id}' has task " f"'{self.task}'; audiobook.py can only synthesize with TTS " f"model entries (tasks {', '.join(AUDIOCPP_SYNTHESIS_TASKS)}). " f"Pick a synthesis entry with --model (available: {available})." ) def _unload_server_models(self) -> None: """Ask the server to unload every loaded model before generating. Lazy-loaded entries stay resident until the server exits (unless its max_loaded_models setting bounds residency), so switching between configured models across runs can exhaust device memory. Unloading first frees those leftovers; this run's model reloads transparently on its first request. Failures only warn: an older server without the endpoint, or a busy one, must not block a working setup. Controlled by config.AUDIOCPP_UNLOAD_MODELS (the TUI Settings "Unload models" option). """ request = urllib.request.Request( f"{self.api_url}/v1/tasks/unload_all_models", data=b"", method="POST", headers={"Content-Type": "application/json"}) try: with urllib.request.urlopen(request, timeout=10) as response: payload = json.loads(response.read().decode("utf-8")) except Exception as exc: self._report(f"[WARNING] Could not unload previously loaded models at " f"{self.api_url}: {exc}") return unloaded = [entry for entry in (payload.get("unloaded") or []) if isinstance(entry, str)] if unloaded: self._report(f"[OK] Unloaded {len(unloaded)} model(s) from server memory: " f"{', '.join(unloaded)}") else: logger.debug("No loaded audio.cpp models to unload at %s", self.api_url) def _get_json(self, path: str, timeout: int = 10) -> Dict[str, Any]: """GET a JSON document from the server.""" url = f"{self.api_url}{path}" try: with urllib.request.urlopen(url, timeout=timeout) as response: return json.loads(response.read().decode("utf-8")) except urllib.error.HTTPError as exc: detail = "" try: detail = exc.read().decode("utf-8", errors="replace")[:200] except Exception: pass raise RuntimeError( f"audio.cpp server returned HTTP {exc.code} for {path}: {detail}") from exc except urllib.error.URLError as exc: raise RuntimeError(f"audio.cpp request failed for {path}: {exc.reason}") from exc def _check_health(self) -> None: """Verify the server is reachable and reports healthy.""" try: payload = self._get_json("/health") except Exception as exc: raise RuntimeError( f"audio.cpp server not reachable at {self.api_url}: {exc}. " "Start audiocpp_server first (see the 'audio.cpp backend' " "section of the README)." ) from exc if payload.get("status") != "ok": raise RuntimeError( f"The audio.cpp server at {self.api_url} reports status " f"{payload.get('status')!r} instead of 'ok'") def _list_models(self) -> List[Dict[str, str]]: """Fetch the (id, family, task) triples reported by the server.""" try: payload = self._get_json("/v1/models") except Exception as exc: raise RuntimeError( f"The audio.cpp server at {self.api_url} did not answer " f"/v1/models: {exc}") from exc entries = payload.get("data") or [] models: List[Dict[str, str]] = [] for entry in entries: if isinstance(entry, dict) and entry.get("id"): models.append({ "id": entry["id"], "family": entry.get("family") or "", "task": entry.get("task") or "", }) return models def _auto_pick_model_id(self, models: List[Dict[str, str]]) -> None: """Resolve an empty model id when the server hosts exactly one entry. Multi-model servers generated with several lazily-loaded entries need an explicit ``--model``, since guessing would risk synthesizing a whole book with the wrong family. """ if self.model_id: return if len(models) == 1: self.model_id = models[0]["id"] logger.info( "No --model given; using the only server entry '%s'", self.model_id) else: logger.debug( "No --model given and the server hosts %d entries; " "an explicit --model is required", len(models)) def _require_model_id(self, models: List[Dict[str, str]]) -> None: """Verify the model id chosen for this run exists on the server.""" model_ids = [model["id"] for model in models] if self.model_id and self.model_id in model_ids: return configured = ", ".join(model_ids) or "none" if not self.model_id: raise RuntimeError( f"The audio.cpp server at {self.api_url} hosts {len(model_ids)} " f"model entries ({configured}); audiobook.py needs to know which " "one to use. Pass --model when converting (see README)." ) if self.preset_mode: raise RuntimeError( f"The audio.cpp server at {self.api_url} has no model id " f"'{self.model_id}' (configured: {configured}). Pass " "--model naming one of the hosted TTS model entries " "(see README)." ) raise RuntimeError( f"The audio.cpp server at {self.api_url} has no model id " f"'{self.model_id}' (configured: {configured}). Select the " "Qwen3-TTS CustomVoice entry for built-in speakers, or rerun " "with --voice NAME matching a voice_preset or voice_dir entry " "on any TTS model (see README)." ) def _resolve_family(self, models: List[Dict[str, str]]) -> None: """Resolve the selected model's family and its request profile. The family comes from GET /v1/models; a missing family is an unknown family that falls through to the generic (clone-only) profile rather than guessing a specific one — audiocpp_server always reports family for entries its server.json describes. """ entry = next( (model for model in models if model["id"] == self.model_id), None) family = (entry["family"] if entry is not None else "") or "" self.family = family self.profile = AUDIOCPP_FAMILY_PROFILES.get( family, AUDIOCPP_DEFAULT_FAMILY_PROFILE) if not family: logger.debug("Model '%s' reported no family; using the generic " "profile", self.model_id) elif family not in AUDIOCPP_FAMILY_PROFILES: logger.info( "audio.cpp family '%s' has no dedicated profile; using the " "generic profile (voice cloning via --voice, model-detected " "language)", family) def _resolve_task(self, models: List[Dict[str, str]]) -> None: """Resolve the selected model's task (tts, clon, vdes, ...) and set design mode for voice design entries. The task comes from GET /v1/models and is fixed per server entry by its server.json config (a VoiceDesign model must be hosted with "task": "vdes"). Servers that predate the task field hosted plain TTS models, so a missing task is treated as tts. """ entry = next( (model for model in models if model["id"] == self.model_id), None) task = (entry["task"] if entry is not None else "") or "" if not task: task = AUDIOCPP_TASK_TTS logger.debug("Model '%s' reported no task; assuming tts", self.model_id) self.task = task self.design_mode = task == AUDIOCPP_TASK_VDES def _check_voice(self) -> None: """Verify the requested voice is available on the server. A voice name that matches no server preset or voice-library wav would be passed through to the model as a cached voice id; on the Base (cloning) model that is silently ignored and plain TTS audio comes back, so preset names are validated up front. When the voices endpoint cannot be queried, validation is skipped with a warning rather than blocking the run. """ query = urllib.parse.urlencode({"model": self.model_id}) try: payload = self._get_json(f"/v1/audio/voices?{query}") except Exception as exc: logger.warning("Could not list server voices; skipping voice " "validation: %s", exc) return voices = payload.get("voices") or [] if self.voice not in voices: available = ", ".join(str(v) for v in voices) or "none" raise RuntimeError( f"Voice '{self.voice}' is not available on the audio.cpp server " f"(available: {available}). Configure it as a voice_preset or " "voice_dir entry in the server config, or pass a listed name " "with --voice (see README)." ) # ------------------------------------------------------------------ # HTTP requests # ------------------------------------------------------------------ def _request_wav(self, text: str) -> bytes: """POST one sub-chunk and return the raw WAV bytes.""" url = f"{self.api_url}/v1/audio/speech" payload: Dict[str, Any] = { "model": self.model_id, "input": text, } # Design models take no voice field (the voice comes from the # instruction); instruction-voice runs on families without built-in # speakers omit it too, since no speaker or preset was requested. if not self.design_mode and not self.instruction_voice: payload["voice"] = self.voice if self.profile.language_style == AUDIOCPP_LANG_DISPLAY: payload["language"] = self.language elif self.profile.language_style == AUDIOCPP_LANG_ISO: iso_code = LANGUAGE_ISO_CODES.get(self.language) if iso_code: payload["language"] = iso_code else: # "Auto": no code to send, so let the server pick its default. logger.debug("%s: no language code for %r; omitted from request", self.family, self.language) if self._seed >= 0: # audio.cpp has no negative "randomize" seed; a negative seed # means "let the server randomize", so the field is omitted. payload["seed"] = self._seed if self.instructions: # Explicit voice-design or style instruction (required for task # "vdes" entries; a Ctrl/style control on families that read it). payload["instructions"] = self.instructions if self.request_options: # Generic per-model controls (--option KEY=VALUE): forwarded # verbatim; the model ignores keys it does not know. payload["options"] = dict(self.request_options) request = urllib.request.Request( url, data=json.dumps(payload).encode("utf-8"), headers={"Content-Type": "application/json"}, method="POST") timeout = config.API_TIMEOUT try: with urllib.request.urlopen(request, timeout=timeout) as response: wav = response.read() except urllib.error.HTTPError as exc: detail = "" try: detail = exc.read().decode("utf-8", errors="replace")[:200] except Exception: pass raise audiocpp_request_error(exc.code, detail, voice=self.voice) from exc except urllib.error.URLError as exc: raise RuntimeError(f"audio.cpp request failed: {exc.reason}") from exc if len(wav) < 12 or wav[:4] != b"RIFF" or wav[8:12] != b"WAVE": raise RuntimeError("audio.cpp server returned audio that is not a WAV file") return wav # ------------------------------------------------------------------ # Chunk generation # ------------------------------------------------------------------ def generate_chunk(self, text: str, chunk_num: int) -> Optional[str]: """Generate one audio chunk; returns its path in the chunks folder. The text is split into sub-requests of at most ``config.CHUNK_SIZE`` words each; each sub-request returns a complete WAV file and the parts are concatenated into one chunk file. """ try: sub_texts = split_into_chunks(text, max_words=config.CHUNK_SIZE) if not sub_texts: raise RuntimeError("No text to synthesize") output_path: Optional[Path] = None with tempfile.TemporaryDirectory(prefix="tts_parts_") as parts_dir, \ self._chunk_heartbeat(chunk_num): part_paths = [] for sub_num, sub_text in enumerate(sub_texts, 1): wav = self._request_wav(sub_text) destination = Path(parts_dir) / f"part_{sub_num:02d}.wav" destination.write_bytes(wav) part_paths.append(destination) if len(part_paths) == 1: output_path = self._chunk_path(chunk_num, ".wav") shutil.copy2(part_paths[0], output_path) else: output_path = self._chunk_path(chunk_num, ".wav") concat_audio_files(part_paths, output_path) logger.debug("Chunk %d generated successfully (%d sub-request(s))", chunk_num, len(sub_texts)) return str(output_path) except ConversionCancelled: raise except NonRetryableTTSError: # Propagate past the generic handler so the retry loop skips # its remaining attempts for deterministic server errors. raise except Exception as exc: logger.error("audio.cpp chunk processing failed for chunk %d: %s", chunk_num, exc) return None