aboutsummaryrefslogtreecommitdiff
path: root/app/converter/tts.py
diff options
context:
space:
mode:
authorhistoria <historiavg@proton.me>2026-08-26 01:43:41 -0400
committerhistoria <historiavg@proton.me>2026-08-26 01:43:41 -0400
commitacbd9ff2c91182d96c57ffb57bee6e9b3fcbcbd4 (patch)
treee336c11f2a57cff5566e249aa5d5477a3dc63c55 /app/converter/tts.py
parent104a0d65c1ba37847c15b64212b7fec8ba371ccb (diff)
downloadtts-audiobook-generator-acbd9ff2c91182d96c57ffb57bee6e9b3fcbcbd4.tar.gz
refactor: split tts.py into per-backend packages
Diffstat (limited to 'app/converter/tts.py')
-rw-r--r--app/converter/tts.py1469
1 files changed, 0 insertions, 1469 deletions
diff --git a/app/converter/tts.py b/app/converter/tts.py
deleted file mode 100644
index 8130a44..0000000
--- a/app/converter/tts.py
+++ /dev/null
@@ -1,1469 +0,0 @@
-"""Client wrappers for the TTS backends.
-
-QwenTTSClient talks to the Qwen3-TTS demo server (custom voice / voice clone).
-FasterTTSClient talks to the OpenAI-compatible server from the
-faster-qwen3-tts repository (voice cloning only; the reference voice is
-configured server-side — see the "Faster backend" section of the README).
-AudioCppTTSClient talks to the audiocpp_server from the audio.cpp
-repository, which can host any TTS model family audio.cpp supports
-(Qwen3-TTS, Higgs Audio, VoxCPM2, IndexTTS2, ...) through one OpenAI-style
-API; the family is detected from the server at startup (see the
-"audio.cpp backend" sections of the README).
-"""
-
-import contextlib
-import io
-import json
-import logging
-import random
-import shutil
-import sys
-import tempfile
-import threading
-import time
-import urllib.error
-import urllib.parse
-import urllib.request
-import wave
-from pathlib import Path
-from typing import Any, Dict, List, Optional, Tuple
-
-from . import config
-from .audio import concat_audio_files
-from .chunking import split_into_chunks
-
-logger = logging.getLogger(__name__)
-
-
-class ConversionCancelled(Exception):
- """Raised inside a conversion whose cancel event was set.
-
- The TUI run view sets a ``threading.Event`` on the TTS client (and the
- converter checks it between chunks/chapters/books); the retry loops
- raise this so the cancellation propagates out of a sleeping or retrying
- request promptly instead of finishing the retry ladder.
- """
-
-
-# Voice modes (re-exported for the CLI and the converter orchestrator).
-VOICE_MODE_CUSTOM = "custom_voice"
-VOICE_MODE_CLONE = "voice_clone"
-VOICE_MODES = (VOICE_MODE_CUSTOM, VOICE_MODE_CLONE)
-
-# TTS backends (re-exported for the CLI and the converter orchestrator).
-BACKEND_QWEN = "qwen"
-BACKEND_FASTER = "faster"
-BACKEND_AUDIOCPP = "audiocpp"
-BACKENDS = (BACKEND_AUDIOCPP, BACKEND_QWEN, BACKEND_FASTER)
-
-# Languages understood by the Qwen3-TTS API. Display names must match the
-# demo dropdown exactly (the demo silently falls back to "Auto" for
-# unrecognized values, so languages are validated client-side first).
-TTS_LANGUAGES = (
- "Auto",
- "Chinese",
- "English",
- "German",
- "Italian",
- "Portuguese",
- "Spanish",
- "Japanese",
- "Korean",
- "French",
- "Russian",
-)
-
-# Short aliases accepted on the command line (ISO 639-1 codes and common
-# shorthands), mapped to the display names above.
-TTS_LANGUAGE_ALIASES = {
- "zh": "Chinese",
- "en": "English",
- "de": "German",
- "it": "Italian",
- "pt": "Portuguese",
- "es": "Spanish",
- "ja": "Japanese",
- "ko": "Korean",
- "fr": "French",
- "ru": "Russian",
- "zh-cn": "Chinese",
- "zh-tw": "Chinese",
- "pt-br": "Portuguese",
- "en-us": "English",
- "en-gb": "English",
-}
-
-# Qwen display names -> ISO 639-1 codes, for audio.cpp families whose
-# language request option takes a code instead of a display name. "Auto"
-# has no code and maps to None so the field is omitted and the server
-# applies its own default.
-LANGUAGE_ISO_CODES = {
- "Chinese": "zh",
- "English": "en",
- "German": "de",
- "Italian": "it",
- "Portuguese": "pt",
- "Spanish": "es",
- "Japanese": "ja",
- "Korean": "ko",
- "French": "fr",
- "Russian": "ru",
-}
-
-# --- audio.cpp model families ---------------------------------------------
-#
-# audiocpp_server exposes the same OpenAI-style API for every TTS family it
-# hosts; families only differ in a few request conventions, captured here as
-# profiles. Families that are not listed use the default profile below.
-
-# How the "language" request field is expressed by a family.
-AUDIOCPP_LANG_DISPLAY = "display" # Qwen display names, e.g. "English"
-AUDIOCPP_LANG_ISO = "iso" # ISO 639-1 codes, e.g. "en"
-AUDIOCPP_LANG_OMIT = "omit" # no language field; the model detects it
-
-# The Qwen3-TTS family. Unlike every other family (one model type each),
-# qwen3_tts hosts several model *types* under one family id, distinguished
-# only by the server entry's id/task: the CustomVoice model (built-in
-# speakers, e.g. Vivian/Ryan), the Base model (voice cloning via a
-# server-side preset), and the VoiceDesign model (task "vdes"). The
-# per-entry voice capability below (audiocpp_entry_voice_capability)
-# resolves which is which, driving both the Convert form (which voice
-# list to show) and the converter's mode selection.
-AUDIOCPP_FAMILY_QWEN3_TTS = "qwen3_tts"
-
-# Server model entry tasks this client can synthesize audiobooks with,
-# taken from GET /v1/models (the "task" field of each entry; a missing task
-# is treated as "tts" — a harmless generic default). "vdes" entries are
-# voice design models: the voice is described with --instructions instead
-# of coming from a speaker or a reference clip. Entries with any other task
-# (asr, vc, diar, ...) are rejected at connect time with a hint to pick a
-# synthesis entry.
-AUDIOCPP_TASK_TTS = "tts"
-AUDIOCPP_TASK_VDES = "vdes"
-AUDIOCPP_SYNTHESIS_TASKS = (AUDIOCPP_TASK_TTS, "clon", AUDIOCPP_TASK_VDES)
-
-# The voice capability of a server model entry — how its voice is supplied.
-# Resolved per entry from (family, task, id) by
-# audiocpp_entry_voice_capability; drives both the Convert form (which
-# voice list to show) and the converter (speaker vs preset vs design mode).
-# Most families are clone-only; only the Qwen3-TTS CustomVoice entry has
-# built-in speakers, and only VoiceDesign entries take a description.
-AUDIOCPP_VOICE_SPEAKER = "speaker" # built-in speaker name (Qwen CustomVoice)
-AUDIOCPP_VOICE_CLONE = "clone" # server-side preset / voice_dir (Base, others)
-AUDIOCPP_VOICE_DESIGN = "design" # voice described by --instructions (vdes)
-
-
-class AudioCppFamilyProfile:
- """Request conventions of one audio.cpp model family.
-
- Language style and whether the family reads a style/instruction prompt;
- these are family-level (every entry of a family shares them). Whether a
- *specific entry* has built-in speakers is an entry-level concern, decided
- by audiocpp_entry_voice_capability, not this profile.
- """
-
- def __init__(self, language_style: str = AUDIOCPP_LANG_OMIT,
- sends_instructions: bool = False):
- self.language_style = language_style
- self.sends_instructions = sends_instructions
-
-
-# Generic profile for families not listed in AUDIOCPP_FAMILY_PROFILES:
-# clone-only, no style instructions, and no language field (the model
-# detects the language itself). Describes higgs_audio_tts, voxcpm2,
-# fish_audio, dots_tts, dramabox, omnivoice, outetts, glm_tts, miotts,
-# moss_tts_*, pocket_tts, vibevoice, ... as well as families added to
-# audio.cpp after this table was written.
-AUDIOCPP_DEFAULT_FAMILY_PROFILE = AudioCppFamilyProfile()
-
-AUDIOCPP_FAMILY_PROFILES = {
- AUDIOCPP_FAMILY_QWEN3_TTS: AudioCppFamilyProfile(
- language_style=AUDIOCPP_LANG_DISPLAY,
- sends_instructions=True,
- ),
- # Families whose language option takes a code (e.g. "en") instead of
- # a Qwen display name; otherwise clone-only like the default profile.
- "chatterbox": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO),
- "confucius4_tts": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO),
- "index_tts2": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO),
- "magpie_tts": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO),
- "supertonic": AudioCppFamilyProfile(language_style=AUDIOCPP_LANG_ISO),
-}
-
-
-def audiocpp_entry_voice_capability(family: str, task: str,
- model_id: str) -> str:
- """How a server model entry's voice is supplied — speaker/clone/design.
-
- Resolved from the entry's family, task and id — the same {id, family,
- task} triple GET /v1/models reports, so it works for local server.json
- entries and remote live-queried entries alike. Qwen3-TTS is the one
- family hosting several model *types* under one family id: the
- CustomVoice model (id contains "customvoice") has built-in speakers, the
- Base model and any other entry are clone-only, and VoiceDesign entries
- (task "vdes") take a description. Every other family is clone-only.
- """
- if task == AUDIOCPP_TASK_VDES:
- return AUDIOCPP_VOICE_DESIGN
- if family == AUDIOCPP_FAMILY_QWEN3_TTS \
- and "customvoice" in (model_id or "").lower():
- return AUDIOCPP_VOICE_SPEAKER
- return AUDIOCPP_VOICE_CLONE
-
-# Built-in CustomVoice speaker names for the Qwen3-TTS family. Shared by the
-# qwen-tts demo backend (config.SPEAKER, the qwen setup/form) and the
-# audio.cpp audiocpp backend's CustomVoice entry (the Convert form's Speaker
-# picker). Entries are the canonical/config form; speaker_display_name()
-# maps them to the wire (display) form via SPEAKER_DISPLAY_NAMES below.
-QWEN3_TTS_SPEAKERS = ("Vivian", "Serena", "Uncle_Fu", "Dylan", "Eric",
- "Ryan", "Aiden", "Ono_Anna", "Sohee")
-
-# Canonical speaker names -> display names used by the qwen-tts demo.
-SPEAKER_DISPLAY_NAMES = {
- "ryan": "Ryan",
- "serena": "Serena",
- "vivian": "Vivian",
- "uncle_fu": "Uncle Fu",
- "aiden": "Aiden",
- "ono_anna": "Ono Anna",
- "sohee": "Sohee",
- "eric": "Eric",
- "dylan": "Dylan",
-}
-
-# Fixed model facts: both demos run the 1.7B model (the CustomVoice demo
-# takes its full HuggingFace id), and the 12Hz codec outputs 24 kHz audio.
-MODEL_SIZE = "1.7B"
-CUSTOM_VOICE_MODEL_ID = "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice"
-SAMPLE_RATE = 24000
-
-CHUNKS_FOLDER = Path(__file__).resolve().parent.parent.parent / "app" / "chunks"
-
-
-def _resolve_request_seed() -> int:
- """Resolve the seed sent with every request.
-
- Returns config.SEED as-is, or (with CONSTANT_SEED and SEED < 0) one
- random value drawn per run, meant to be reused for every request so
- the voice stays consistent across chunk boundaries. Without
- CONSTANT_SEED, -1 is returned so the server re-samples the voice on
- every generation.
- """
- seed = config.SEED
- if config.CONSTANT_SEED and seed < 0:
- seed = random.randrange(2 ** 31)
- return seed
-
-
-def speaker_display_name_for(name: str) -> str:
- """Return the wire (display) form of a Qwen3-TTS CustomVoice speaker NAME.
-
- Accepts either the canonical/config form (e.g. "uncle_fu", "Uncle_Fu")
- or the display form ("Uncle Fu"), case-insensitively; unknown names pass
- through unchanged. Used by AudioCppTTSClient to normalize the --voice /
- Speaker-picker value into what audiocpp_server expects in the request's
- voice field.
- """
- return SPEAKER_DISPLAY_NAMES.get((name or "").lower(), name)
-
-
-def is_builtin_speaker(name: Optional[str]) -> bool:
- """True when NAME is one of the Qwen3-TTS CustomVoice built-in speakers.
-
- Matches case-insensitively across the canonical ("Uncle_Fu"), display
- ("Uncle Fu") and shorthand ("uncle_fu") forms, so the --voice flag and
- the Convert form's Speaker picker resolve to the same set.
- """
- if not name:
- return False
- norm = name.lower().replace("_", " ").replace("-", " ")
- return any(norm == speaker.lower().replace("_", " ")
- for speaker in QWEN3_TTS_SPEAKERS)
-
-
-def speaker_display_name() -> str:
- """Return the display name for the configured custom speaker."""
- return speaker_display_name_for(config.SPEAKER)
-
-
-def normalize_language(value: Optional[str]) -> str:
- """Normalize a user-provided language name to a Qwen3-TTS display name.
-
- Accepts the display names in TTS_LANGUAGES case-insensitively as
- well as the short aliases in TTS_LANGUAGE_ALIASES (ISO 639-1 codes
- and common shorthands). Raises ValueError for anything else, since the
- Qwen3-TTS demo silently falls back to "Auto" for unrecognized languages.
- """
- if value is None:
- raise ValueError("Language must not be None")
- candidate = value.strip()
- if not candidate:
- raise ValueError("Language must not be empty")
- for name in TTS_LANGUAGES:
- if candidate.lower() == name.lower():
- return name
- alias = TTS_LANGUAGE_ALIASES.get(candidate.lower())
- if alias:
- return alias
- raise ValueError(
- f"Unknown language: {value!r}. Expected one of "
- f"{', '.join(TTS_LANGUAGES)} (or an alias: "
- f"{', '.join(sorted(TTS_LANGUAGE_ALIASES))})."
- )
-
-
-def transcribe_reference_audio(audio_path: str, model_name: str = "base") -> Optional[str]:
- """Transcribe reference audio locally using an optional Whisper backend.
-
- The current qwen-tts demo does not expose a transcription endpoint, so
- transcription is done client-side when a Whisper package is available.
- Returns None if no backend is installed.
- """
- for backend in ("faster_whisper", "whisper"):
- try:
- if backend == "faster_whisper":
- from faster_whisper import WhisperModel
- model = WhisperModel(model_name, device="cpu", compute_type="int8")
- segments, _ = model.transcribe(audio_path)
- text = " ".join(seg.text.strip() for seg in segments).strip()
- else:
- import whisper
- model = whisper.load_model(model_name)
- result = model.transcribe(audio_path)
- text = (result.get("text") or "").strip()
- if text:
- logger.info("Transcription complete via %s: %s", backend, text)
- return text
- except ImportError:
- continue
- except Exception as exc:
- logger.warning("%s transcription failed: %s", backend, exc)
- logger.warning("No Whisper backend available; transcription skipped.")
- return None
-
-
-def whisper_backend_available() -> Optional[str]:
- """Return the name of an importable Whisper backend, or None.
-
- Checks faster_whisper first (preferred), then the openai-whisper
- package, without importing the heavy model code: a bare import probe
- is enough to tell whether the package is installed in the current
- environment. Used by the make_audiocpp_server_json tool to warn when
- neither is present (e.g. the wrong conda environment is active).
- """
- for backend in ("faster_whisper", "whisper"):
- try:
- __import__(backend)
- except ImportError:
- continue
- return backend
- return None
-
-
-class _BaseTTSClient:
- """Shared chunk retry logic, heartbeat, and chunk file bookkeeping."""
-
- # Set by the converter when the run is cancellable (the TUI run view):
- # a threading.Event that, once set, aborts the run between requests
- # (and interrupts retry back-off sleeps). ``quiet`` silences console
- # prints (the run view owns the screen).
- cancel = None
- quiet = False
-
- def _report(self, message: str) -> None:
- """Print a console line unless quiet (the run view owns the screen)."""
- if not self.quiet:
- print(message)
-
- def generate_chunk(self, text: str, chunk_num: int) -> Optional[str]:
- """Generate one audio chunk; returns its path in the chunks folder."""
- raise NotImplementedError
-
- def _cancel_requested(self) -> bool:
- """True when the run's cancel event has been set (if any)."""
- return isinstance(self.cancel, threading.Event) \
- and self.cancel.is_set()
-
- def _check_cancelled(self) -> None:
- """Raise ConversionCancelled when the cancel event is set."""
- if self._cancel_requested():
- raise ConversionCancelled("Cancelled by user")
-
- def _sleep(self, seconds: float) -> None:
- """Sleep SECONDS, cut short (raising) when the cancel event sets."""
- if isinstance(self.cancel, threading.Event):
- if self.cancel.wait(seconds):
- raise ConversionCancelled("Cancelled by user")
- else:
- time.sleep(seconds)
-
- def _chunk_path(self, chunk_num: int, suffix: str) -> Path:
- """Resolve the target path for a chunk, removing stale files first.
-
- Any stale chunk file for this index is removed so a retry or extension
- change can never leave two files matching chunk_NNNN.*.
- """
- for stale in CHUNKS_FOLDER.glob(f"chunk_{chunk_num:04d}.*"):
- try:
- stale.unlink()
- except OSError as exc:
- logger.debug("Could not remove stale chunk file %s: %s", stale, exc)
- return CHUNKS_FOLDER / f"chunk_{chunk_num:04d}{suffix}"
-
- def process_chunk_with_retry(self, chunk_num: int, text: str) -> Optional[Path]:
- """Process a chunk with retry logic.
-
- Returns the generated chunk file's path, or None when all attempts
- failed. Raises ConversionCancelled when the run was cancelled.
- """
- for attempt in range(config.MAX_RETRIES):
- self._check_cancelled()
- try:
- result = self.generate_chunk(text, chunk_num)
- if result and Path(result).exists():
- return Path(result)
- logger.warning("Chunk %d attempt %d failed", chunk_num, attempt + 1)
- except ConversionCancelled:
- raise
- except Exception as exc:
- logger.warning("Chunk %d attempt %d error: %s", chunk_num, attempt + 1, exc)
-
- if attempt < config.MAX_RETRIES - 1:
- sleep_time = 5 + (2 ** attempt)
- logger.info("Waiting %ds before retry...", sleep_time)
- self._sleep(sleep_time)
-
- logger.error("Chunk %d failed after %d attempts", chunk_num, config.MAX_RETRIES)
- return None
-
- @contextlib.contextmanager
- def _chunk_heartbeat(self, chunk_num: int):
- """Log a periodic "still working" record while a request generates."""
- stop = threading.Event()
- subject = f"Chunk {chunk_num}"
-
- def _beat():
- start = time.time()
- while not stop.wait(config.HEARTBEAT_INTERVAL_SECONDS):
- elapsed = time.time() - start
- if self.quiet:
- logger.info("%s still generating — %dm %ds elapsed",
- subject, int(elapsed // 60), int(elapsed % 60))
- else:
- print(f"[...] {subject} still generating — "
- f"{int(elapsed // 60)}m {int(elapsed % 60)}s elapsed",
- flush=True)
-
- thread = threading.Thread(target=_beat, daemon=True)
- thread.start()
- try:
- yield
- finally:
- stop.set()
- thread.join()
-
-
-class QwenTTSClient(_BaseTTSClient):
- """Generates audio chunks through a Qwen3-TTS demo server."""
-
- def __init__(self, voice_mode: str = "custom_voice", voice_clone_ref_audio: Optional[str] = None,
- voice_clone_ref_text: Optional[str] = None, skip_transcription: bool = False,
- language: Optional[str] = None, api_url: Optional[str] = None,
- quiet: bool = False):
- # Quiet before connecting so connect-time status lines never reach
- # a screen the TUI run view owns.
- self.quiet = bool(quiet)
- if voice_mode not in VOICE_MODES:
- raise ValueError(
- f"Unknown voice mode: {voice_mode!r} (expected one of {VOICE_MODES})"
- )
- self.voice_mode = voice_mode
- self.voice_clone_ref_audio = voice_clone_ref_audio
- self.voice_clone_ref_text = (voice_clone_ref_text or "").strip()
- self.skip_transcription = skip_transcription
- # api_url overrides the configured endpoint for the active voice mode
- # (used by the hub's "[remote]" backend entries and --api-url).
- self.api_url = (api_url or "").strip() or None
- # Seed sent with every request: config.SEED as-is, or (with
- # CONSTANT_SEED and SEED < 0) one random value drawn per run and
- # reused for every request so the voice stays consistent across
- # chunk boundaries. Without CONSTANT_SEED, -1 is forwarded so the
- # server re-samples the voice on every generation.
- self._seed = _resolve_request_seed()
- if language is None:
- language = config.LANGUAGE
- # Validate before connecting so bad values fail fast without a server.
- self.language = normalize_language(language)
- self.client = None
- self.api_info: Dict[str, Any] = {}
- self.clone_client = None
- self.clone_api_info: Dict[str, Any] = {}
- self._ref_audio_filedata: Optional[Dict[str, Any]] = None
- self._connect()
-
- # ------------------------------------------------------------------
- # Connection
- # ------------------------------------------------------------------
-
- def _connect(self) -> None:
- api_url = self.api_url or (
- config.CLONE_API_URL if self.voice_mode == VOICE_MODE_CLONE
- else config.QWEN_API_URL)
- try:
- if self.voice_mode == VOICE_MODE_CLONE:
- # Voice clone uses the Base-model demo, which is a separate server
- # from the CustomVoice demo (that one only exposes /run_instruct).
- self._init_client(api_url, clone=True)
- self._report(f"[OK] Connected to Voice Clone API at {api_url}")
- self._resolve_reference_text()
- else:
- self._init_client(api_url, clone=False)
- self._report("[OK] Connected to Qwen API")
- except Exception as exc:
- raise RuntimeError(
- f"Qwen API initialization failed at {api_url}: {exc}. "
- "Make sure the Qwen demo server is running and reachable, and that your "
- "installed Qwen3-TTS version matches this converter's API expectations "
- "(voice clone requires the Base-model demo: Qwen/Qwen3-TTS-12Hz-1.7B-Base)."
- ) from exc
-
- def _resolve_reference_text(self) -> None:
- """Resolve the reference transcript: explicit text, then local
- transcription, then x-vector-only mode."""
- if not self.voice_clone_ref_text and self.voice_clone_ref_audio:
- if self.skip_transcription:
- self._report("[INFO] Skipping reference audio transcription (--no-transcription).")
- else:
- self._report("[INFO] Transcribing reference audio for voice cloning...")
- self.voice_clone_ref_text = self.transcribe_audio(self.voice_clone_ref_audio) or ""
- if not self.voice_clone_ref_text:
- self._report("[WARNING] No reference text available; using "
- "x-vector-only clone mode (lower quality).")
- self._report(' Pass --transcription "..." for higher-quality in-context cloning.')
- else:
- self._report(f"[OK] Reference text:\n{self.voice_clone_ref_text}")
-
- def _init_client(self, url: str, clone: bool = False) -> None:
- """Initialize a Gradio client and store its API metadata.
-
- gradio_client prints its usage info directly to stdout while the
- client is created and its API metadata loaded, so stdout is swapped
- for a buffer for the whole process; the captured text is re-emitted
- at DEBUG level for troubleshooting.
- """
- from gradio_client import Client
-
- logger.info("Connecting to Qwen API at %s...", url)
- old_stdout = sys.stdout
- captured = io.StringIO()
- sys.stdout = captured
- try:
- try:
- client = Client(url, httpx_kwargs={"timeout": config.API_TIMEOUT})
- except TypeError:
- # Older gradio_client versions don't support httpx_kwargs.
- client = Client(url)
- if clone:
- self.clone_client = client
- self.clone_api_info = self._load_api_info(client)
- else:
- self.client = client
- self.api_info = self._load_api_info(client)
- finally:
- sys.stdout = old_stdout
- usage_info = captured.getvalue().strip()
- if usage_info:
- logger.debug("Gradio client output for %s:\n%s", url, usage_info)
- logger.info("Connected to Qwen API")
-
- @staticmethod
- def _load_api_info(client) -> Dict[str, Any]:
- """Load available API metadata from the Gradio app."""
- try:
- return client.view_api(return_format="dict")
- except Exception as exc:
- logger.warning("Unable to read API metadata: %s", exc)
- return {}
-
- def _resolve_api_name(self, *candidates: str, api_info: Optional[Dict[str, Any]] = None) -> str:
- """Return the first available api_name from candidate list."""
- info = api_info if api_info is not None else self.api_info
- named_endpoints = info.get("named_endpoints", {})
- for candidate in candidates:
- if candidate in named_endpoints:
- return candidate
- return candidates[0]
-
- def _endpoint_accepts_param(self, api_name: str, param_name: str,
- api_info: Optional[Dict[str, Any]] = None) -> bool:
- """Check whether endpoint input schema includes the given parameter."""
- info = api_info if api_info is not None else self.api_info
- endpoint = info.get("named_endpoints", {}).get(api_name, {})
- parameters = endpoint.get("parameters", [])
- return any(parameter.get("parameter_name") == param_name for parameter in parameters)
-
- # ------------------------------------------------------------------
- # Reference audio transcription (voice clone)
- # ------------------------------------------------------------------
-
- def transcribe_audio(self, audio_path: str) -> Optional[str]:
- """Transcribe reference audio locally using an optional Whisper backend."""
- return transcribe_reference_audio(audio_path)
-
- # ------------------------------------------------------------------
- # Chunk generation
- # ------------------------------------------------------------------
-
- def generate_chunk(self, text: str, chunk_num: int) -> Optional[str]:
- """Generate one audio chunk; returns its path in the chunks folder.
-
- The text is split into sub-requests of at most
- ``config.CHUNK_SIZE`` words each (the book-level chunker
- normally guarantees this already; the split is defense in depth
- against pathological input such as a punctuation-free run of
- text), and the audio files returned for the sub-requests are
- concatenated into one chunk file.
- """
- try:
- sub_texts = split_into_chunks(text, max_words=config.CHUNK_SIZE)
- if not sub_texts:
- raise RuntimeError("No text to synthesize")
-
- output_path: Optional[Path] = None
- with tempfile.TemporaryDirectory(prefix="tts_parts_") as parts_dir, \
- self._chunk_heartbeat(chunk_num):
- part_paths = [
- self._generate_sub_request(sub_text, parts_dir, sub_num,
- len(sub_texts), chunk_num)
- for sub_num, sub_text in enumerate(sub_texts, 1)
- ]
- if len(part_paths) == 1:
- suffix = part_paths[0].suffix or ".wav"
- output_path = self._chunk_path(chunk_num, suffix)
- shutil.copy2(part_paths[0], output_path)
- else:
- output_path = self._chunk_path(chunk_num, ".wav")
- concat_audio_files(part_paths, output_path)
-
- logger.debug("Chunk %d generated successfully (%d sub-request(s))",
- chunk_num, len(sub_texts))
- return str(output_path)
-
- except ConversionCancelled:
- raise
- except Exception as exc:
- logger.error("Qwen chunk processing failed for chunk %d: %s", chunk_num, exc)
- return None
-
- def _generate_sub_request(self, text: str, parts_dir: str, sub_num: int,
- sub_total: int, chunk_num: int) -> Path:
- """Run one API generation for ``text``; returns the downloaded audio."""
- if sub_total > 1:
- logger.info("Chunk %d: oversized input split into %d requests "
- "(sub-request %d/%d)", chunk_num, sub_total, sub_num, sub_total)
- if self.voice_mode == VOICE_MODE_CUSTOM:
- result = self._generate_custom_voice(text)
- elif self.voice_mode == VOICE_MODE_CLONE:
- result = self._generate_voice_clone(text)
- else:
- raise ValueError(f"Unknown voice mode: {self.voice_mode}")
-
- if not isinstance(result, (tuple, list)) or not result:
- raise RuntimeError("Qwen API returned an invalid result")
-
- audio_path = result[0] # First element is the audio file path
- if not isinstance(audio_path, (str, Path)) or not audio_path:
- raise RuntimeError("Qwen API did not return an audio file path")
-
- source = Path(audio_path)
- if not source.exists():
- raise RuntimeError(f"Generated audio file not found: {audio_path}")
-
- destination = Path(parts_dir) / f"part_{sub_num:02d}{source.suffix or '.wav'}"
- shutil.copy2(source, destination)
-
- return destination
-
- # ------------------------------------------------------------------
- # API payloads
- # ------------------------------------------------------------------
-
- def _generate_custom_voice(self, text: str) -> Tuple:
- """Generate audio using CustomVoice mode."""
- custom_api = self._resolve_api_name("/run_instruct", "/run_custom_voice", "/generate_custom_voice")
- if custom_api == "/run_instruct":
- payload = dict(
- text=text,
- lang_disp=self.language,
- spk_disp=speaker_display_name(),
- instruct=config.INSTRUCT,
- )
- else:
- payload = dict(
- text=text,
- language=self.language,
- speaker=config.SPEAKER,
- instruct=config.INSTRUCT,
- )
- if self._endpoint_accepts_param(custom_api, "model_id_cv"):
- payload["model_id_cv"] = CUSTOM_VOICE_MODEL_ID
- elif self._endpoint_accepts_param(custom_api, "model_size"):
- payload["model_size"] = MODEL_SIZE
-
- if self._endpoint_accepts_param(custom_api, "seed"):
- payload["seed"] = self._seed
-
- return self.client.predict(**payload, api_name=custom_api)
-
- def _ref_audio_payload(self) -> Dict[str, Any]:
- """Gradio file payload for the reference audio (built once, reused)."""
- if self._ref_audio_filedata is None:
- from gradio_client import handle_file
- self._ref_audio_filedata = handle_file(self.voice_clone_ref_audio)
- return self._ref_audio_filedata
-
- def _generate_voice_clone(self, text: str) -> Tuple:
- """Generate audio using Voice Clone mode."""
- if not Path(self.voice_clone_ref_audio).exists():
- raise FileNotFoundError(f"Reference audio not found: {self.voice_clone_ref_audio}")
-
- if self.clone_client is None:
- raise RuntimeError("Voice Clone client is not initialized. Is the Base-model demo running?")
-
- clone_api = self._resolve_api_name("/run_voice_clone", "/generate_voice_clone",
- api_info=self.clone_api_info)
- use_xvector = config.XVECTOR_ONLY or not self.voice_clone_ref_text
-
- if clone_api == "/run_voice_clone":
- payload = dict(
- ref_aud=self._ref_audio_payload(),
- ref_txt=self.voice_clone_ref_text,
- use_xvec=use_xvector,
- text=text,
- lang_disp=self.language,
- )
- else:
- payload = dict(
- ref_audio=self._ref_audio_payload(),
- ref_text=self.voice_clone_ref_text,
- target_text=text,
- language=self.language,
- use_xvector_only=use_xvector,
- )
- optional_params = {
- "model_size": MODEL_SIZE,
- "seed": self._seed,
- }
- for name, value in optional_params.items():
- if self._endpoint_accepts_param(clone_api, name, api_info=self.clone_api_info):
- payload[name] = value
-
- return self.clone_client.predict(**payload, api_name=clone_api)
-
-
-class FasterTTSClient(_BaseTTSClient):
- """Generates audio chunks through a faster-qwen3-tts server.
-
- Talks to the OpenAI-compatible server shipped in the faster-qwen3-tts
- repository (examples/openai_server.py). The reference voice (ref audio,
- ref text) and language are configured on the server itself via
- --ref-audio/--ref-text or a --voices JSON file; this client only sends
- text. Unlike the Qwen demo, the server performs one generation per
- request, so long chunks are sub-chunked client-side.
- """
-
- def __init__(self, voice: Optional[str] = None, api_url: Optional[str] = None,
- quiet: bool = False):
- # Quiet before connecting so connect-time status lines never reach
- # a screen the TUI run view owns.
- self.quiet = bool(quiet)
- self.voice = voice or config.FASTER_VOICE
- self.api_url = (api_url or config.FASTER_API_URL).rstrip("/")
- self._check_health()
-
- def _check_health(self) -> None:
- """Verify the server is reachable and its model is loaded."""
- url = f"{self.api_url}/health"
- try:
- with urllib.request.urlopen(url, timeout=10) as response:
- payload = json.loads(response.read().decode("utf-8"))
- except Exception as exc:
- raise RuntimeError(
- f"Faster TTS server not reachable at {url}: {exc}. "
- "Start the faster-qwen3-tts OpenAI-compatible server first "
- "(see the 'Faster backend' section of the README)."
- ) from exc
- if not payload.get("model_loaded"):
- raise RuntimeError(
- "The faster TTS server is running but its model is not loaded yet; "
- "wait for model download and startup to finish, then retry."
- )
- self._report(f"[OK] Connected to faster TTS API at {self.api_url} (voice '{self.voice}')")
- self._report(f"[INFO] The server silently falls back to its first configured voice if "
- f"'{self.voice}' is not defined in its voice config (see README).")
-
- # ------------------------------------------------------------------
- # HTTP requests
- # ------------------------------------------------------------------
-
- def _request_pcm(self, text: str) -> bytes:
- """POST one sub-chunk and return raw 16-bit mono PCM bytes."""
- url = f"{self.api_url}/v1/audio/speech"
- payload = json.dumps({
- "model": "tts-1",
- "input": text,
- "voice": self.voice,
- "response_format": "pcm",
- }).encode("utf-8")
- request = urllib.request.Request(
- url, data=payload, headers={"Content-Type": "application/json"}, method="POST")
- try:
- with urllib.request.urlopen(request, timeout=config.API_TIMEOUT) as response:
- pcm = response.read()
- except urllib.error.HTTPError as exc:
- detail = ""
- try:
- detail = exc.read().decode("utf-8", errors="replace")[:200]
- except Exception:
- pass
- raise RuntimeError(f"Faster TTS server returned HTTP {exc.code}: {detail}") from exc
- except urllib.error.URLError as exc:
- raise RuntimeError(f"Faster TTS request failed: {exc.reason}") from exc
- if not pcm:
- raise RuntimeError("Faster TTS server returned empty audio")
- return pcm
-
- # ------------------------------------------------------------------
- # Chunk generation
- # ------------------------------------------------------------------
-
- def generate_chunk(self, text: str, chunk_num: int) -> Optional[str]:
- """Generate one audio chunk; returns its path in the chunks folder."""
- try:
- sub_chunks = split_into_chunks(text, max_words=config.CHUNK_SIZE)
- if not sub_chunks:
- raise RuntimeError("No text to synthesize")
-
- pcm_parts: List[bytes] = []
- with self._chunk_heartbeat(chunk_num):
- for sub_num, sub_text in enumerate(sub_chunks, 1):
- pcm = self._request_pcm(sub_text)
- pcm_parts.append(pcm)
-
- output_path = self._chunk_path(chunk_num, ".wav")
- with wave.open(str(output_path), "wb") as wav_file:
- wav_file.setnchannels(1)
- wav_file.setsampwidth(2)
- wav_file.setframerate(SAMPLE_RATE)
- wav_file.writeframes(b"".join(pcm_parts))
-
- logger.debug("Chunk %d generated (%d sub-chunks)", chunk_num, len(sub_chunks))
- return str(output_path)
-
- except ConversionCancelled:
- raise
- except Exception as exc:
- logger.error("Faster chunk processing failed for chunk %d: %s", chunk_num, exc)
- return None
-
-
-class AudioCppTTSClient(_BaseTTSClient):
- """Generates audio chunks through an audio.cpp audiocpp_server.
-
- Talks to the OpenAI-style HTTP API of audiocpp_server, which hosts TTS
- model families through a native ggml runtime (GGUF weights, no Python
- serving stack). The server API is family-agnostic; the family and task
- of the configured model entry are read from GET /v1/models at startup
- and adapt the request payload (language field style, style instructions)
- through AUDIOCPP_FAMILY_PROFILES. The entry's voice capability
- (audiocpp_entry_voice_capability: speaker / clone / design) decides how
- its voice is supplied; all three are resolved server-side from the
- request's "voice"/"instructions" fields:
-
- - Speaker mode (--voice with a built-in speaker name, or no flag on a
- CustomVoice entry): Qwen3-TTS CustomVoice only. A built-in speaker
- name (e.g. "Vivian") is passed through, plus the INSTRUCT style
- prompt. The selected entry must be the CustomVoice model (capability
- == speaker); a speaker name on a non-speaker entry is treated as a
- server-side preset instead.
- - Preset mode (--voice NAME): a voice configured on the server
- (``voice_presets`` or ``voice_dir`` in its config, e.g. a cloning
- reference). The name is validated against GET /v1/audio/voices at
- startup because an unresolvable name would silently fall back to
- plain TTS on a clone-based model instead of failing. When
- AUDIOCPP_CLONE_MODEL_ID names a second server entry of the same
- family (typically the Qwen Base model), preset requests are routed
- to it. Selecting a non-speaker --voice on a CustomVoice primary with
- a clone id configured is the documented way to switch a speaker setup
- to cloning; without a clone id the voice is validated against the
- server's voice library.
- - Voice design (task "vdes" entries, e.g. Qwen3-TTS VoiceDesign): the
- voice is described in natural language through ``instructions``,
- which is required and sent with every request (no ``voice`` field).
- A constant per-run seed keeps the designed voice consistent across
- chunk boundaries.
-
- The entry's capability decides how an explicit --voice is read: on a
- speaker-capable entry a name that matches a built-in speaker selects
- speaker mode, and every other name is a server-side preset. With no
- --voice the entry's capability picks the mode: design entries require
- --instructions; speaker entries use the built-in CustomVoice speaker
- in config.SPEAKER; clone entries (the Base model, and every other
- family) fail fast with a hint to pass --voice, instead of silently
- synthesizing with a random default voice.
-
- ``instructions`` also works on non-design entries, where it acts as a
- generic style/delivery instruction (voice control): families that read
- it (OmniVoice, Qwen3-TTS CustomVoice, ...) shape the voice or delivery
- accordingly, and others ignore it. On instruction-conditioned families
- without built-in speakers it may replace --voice entirely (the
- instruction defines the voice). Extra request options (``--option
- KEY=VALUE``, e.g. emotion, voice_id, speed) are forwarded verbatim in
- the request's "options" object, which is the server's generic
- pass-through for per-model controls.
-
- Chunking: text is split client-side into sub-requests of at most
- config.CHUNK_SIZE words each; each sub-request returns a complete
- WAV file and the parts are concatenated with the same lossless path
- used for the Qwen client.
- """
-
- def __init__(self, voice: Optional[str] = None, language: Optional[str] = None,
- api_url: Optional[str] = None,
- model_id: Optional[str] = None,
- instructions: Optional[str] = None,
- request_options: Optional[Dict[str, str]] = None,
- quiet: bool = False):
- # Quiet before connecting so connect-time status lines never reach
- # a screen the TUI run view owns.
- self.quiet = bool(quiet)
- self.api_url = (api_url or config.AUDIOCPP_API_URL).rstrip("/")
- # Per-run model selection: the --model CLI flag overrides config; an
- # empty value is resolved at connect time when the server hosts exactly
- # one entry, so multi-model servers don't require editing config.py.
- self.model_id = (model_id if model_id is not None
- else config.AUDIOCPP_MODEL_ID) or ""
- self._model_id_explicit = bool(self.model_id)
- # Validate before connecting so bad values fail fast without a server.
- self.language = normalize_language(
- language if language is not None else config.LANGUAGE)
- # One seed value per run, reused for every request (see
- # _resolve_request_seed). Unlike the Qwen demo, audio.cpp has no
- # negative "randomize" seed, so a negative value means "send no seed
- # at all" (see _request_wav) and the server randomizes.
- self._seed = _resolve_request_seed()
- # Voice selection (the --voice name). preset_mode / speaker_mode are
- # resolved in _connect: a --voice that names a built-in CustomVoice
- # speaker on a speaker-capable entry selects speaker mode; every
- # other name (and any name on a clone-capable entry) is a server-side
- # preset. preset_mode gates _select_model's reroute to
- # AUDIOCPP_CLONE_MODEL_ID and the INSTRUCT style-prompt logic. The
- # request's "voice" field (self.voice) is filled in _connect per the
- # mode.
- self.preset_mode = False
- self.speaker_mode = False
- self.voice = voice or None
- # Style/voice-design instruction sent with every request (the CLI
- # --instructions flag overrides AUDIOCPP_INSTRUCTIONS in config.py).
- # For task "vdes" entries it describes the voice to design; for other
- # families it is a generic style instruction when the model reads one.
- self.instructions = (instructions if instructions is not None
- else config.AUDIOCPP_INSTRUCTIONS or "").strip()
- # Free-form per-request options (--option KEY=VALUE) forwarded in the
- # request's "options" object; models ignore keys they don't know.
- self.request_options: Dict[str, str] = dict(request_options or {})
- # Set during _connect: design_mode for "vdes" entries, instruction_voice
- # when a family without built-in speakers gets its voice from the
- # instruction alone (no voice field). self.voice is also finalized
- # there (the speaker/preset name, or config.SPEAKER for the default).
- self.design_mode = False
- self.instruction_voice = False
- # Family and task of the selected model entry and the family's request
- # profile; all are resolved from GET /v1/models during _connect.
- self.family = ""
- self.task = AUDIOCPP_TASK_TTS
- self.profile = AUDIOCPP_DEFAULT_FAMILY_PROFILE
- self._connect()
-
- # ------------------------------------------------------------------
- # Connection
- # ------------------------------------------------------------------
-
- def _connected(self, mode: str) -> None:
- """Report the resolved connection (MODE: speaker/voice/... label)."""
- self._report(f"[OK] Connected to audio.cpp server at {self.api_url} "
- f"(model '{self.model_id}', family '{self.family}', "
- f"{mode})")
-
- def _connect(self) -> None:
- """Health-check the server and resolve the model, family, task, and voice.
-
- The entry's voice capability (audiocpp_entry_voice_capability, from
- family/task/id) plus the caller's --voice/--instructions pick the
- mode. An explicit --voice on a speaker-capable (CustomVoice) entry
- that names a built-in speaker selects speaker mode; every other
- --voice is a server-side preset, validated against the server's
- voice library (and rerouted to AUDIOCPP_CLONE_MODEL_ID when set).
- With no --voice, design entries require --instructions, speaker-
- capable entries use the built-in config.SPEAKER, and clone entries
- fail fast with a hint instead of silently synthesizing with a
- random default voice.
- """
- self._check_health()
- models = self._list_models()
- self._auto_pick_model_id(models)
- if self.voice is not None:
- # Explicit --voice: decide between speaker mode and a server-side
- # preset. A name matching a built-in CustomVoice speaker on a
- # speaker-capable primary selects speaker mode; every other name
- # (and any name when the primary entry is absent) is a preset,
- # validated against the server's voice library and rerouted to
- # AUDIOCPP_CLONE_MODEL_ID when configured.
- primary = next((m for m in models if m["id"] == self.model_id),
- None)
- if primary is not None:
- self._resolve_family(models)
- self._resolve_task(models)
- capability = audiocpp_entry_voice_capability(
- self.family, self.task, self.model_id)
- if capability == AUDIOCPP_VOICE_SPEAKER \
- and is_builtin_speaker(self.voice):
- self._require_synthesis_task(models)
- self.voice = speaker_display_name_for(self.voice)
- self.speaker_mode = True
- self._connected(f"speaker '{self.voice}'")
- if not self.speaker_mode:
- # Server-side preset (--voice): validate it and route to
- # the clone model entry when AUDIOCPP_CLONE_MODEL_ID is set.
- self.preset_mode = True
- self._select_model(models)
- self._require_model_id(models)
- self._resolve_family(models)
- self._resolve_task(models)
- self._require_synthesis_task(models)
- if self.design_mode:
- raise RuntimeError(
- f"--voice cannot be used with the voice design model "
- f"'{self.model_id}': the voice is described by the "
- "--instructions text instead (see README).")
- self._check_voice()
- self._connected(f"voice '{self.voice}'")
- else:
- # No flag: the entry's capability picks the default mode.
- self._require_model_id(models)
- self._resolve_family(models)
- self._resolve_task(models)
- self._require_synthesis_task(models)
- capability = audiocpp_entry_voice_capability(
- self.family, self.task, self.model_id)
- if self.design_mode:
- if not self.instructions:
- raise RuntimeError(
- f"The audio.cpp model '{self.model_id}' (family "
- f"'{self.family}') is a voice design model: pass a "
- "description of the voice to synthesize with, e.g. "
- '--instructions "A warm adult female narrator with a '
- 'British accent" (see README).')
- self._connected("voice design")
- self._report(f"[INFO] Designing the voice from: {self.instructions}")
- elif capability == AUDIOCPP_VOICE_SPEAKER:
- # No flag on a CustomVoice entry: the built-in config.SPEAKER.
- self.voice = speaker_display_name()
- self.speaker_mode = True
- self._connected(f"speaker '{self.voice}'")
- elif self.instructions:
- # Families without built-in speakers can still get their voice
- # from the instruction alone (e.g. OmniVoice voice design).
- self.instruction_voice = True
- self._connected("instruction voice")
- self._report(f"[INFO] Designing the voice from: {self.instructions}")
- else:
- raise RuntimeError(
- f"The audio.cpp model '{self.model_id}' (family "
- f"'{self.family}') has no built-in speakers, so its voice "
- "must come from the server: rerun with --voice NAME "
- "matching a voice_preset or voice_dir entry in the server "
- "config, or describe a voice with --instructions for "
- "families that support it, or select the CustomVoice entry "
- "for built-in speakers (see README).")
- if self.instructions and not self.design_mode and not self.instruction_voice:
- self._report(f"[INFO] Sending instruction with every request: {self.instructions}")
- self._report("[INFO] Its effect (style, emotion, delivery) depends on the "
- "model family; models without instruction support ignore it.")
- if config.AUDIOCPP_UNLOAD_MODELS:
- self._unload_server_models()
-
- def _require_synthesis_task(self, models: List[Dict[str, str]]) -> None:
- """Reject model entries whose task is not a TTS synthesis task."""
- if self.task in AUDIOCPP_SYNTHESIS_TASKS:
- return
- available = ", ".join(model["id"] for model in models) or "none"
- raise RuntimeError(
- f"The audio.cpp model '{self.model_id}' has task "
- f"'{self.task}'; audiobook.py can only synthesize with TTS "
- f"model entries (tasks {', '.join(AUDIOCPP_SYNTHESIS_TASKS)}). "
- f"Pick a synthesis entry with --model (available: {available})."
- )
-
- def _unload_server_models(self) -> None:
- """Ask the server to unload every loaded model before generating.
-
- Lazy-loaded entries stay resident until the server exits (unless its
- max_loaded_models setting bounds residency), so switching between
- configured models across runs can exhaust device memory. Unloading
- first frees those leftovers; this run's model reloads transparently
- on its first request. Failures only warn: an older server without
- the endpoint, or a busy one, must not block a working setup.
- Controlled by config.AUDIOCPP_UNLOAD_MODELS (the TUI Settings
- "Unload models" option).
- """
- request = urllib.request.Request(
- f"{self.api_url}/v1/tasks/unload_all_models", data=b"",
- method="POST", headers={"Content-Type": "application/json"})
- try:
- with urllib.request.urlopen(request, timeout=10) as response:
- payload = json.loads(response.read().decode("utf-8"))
- except Exception as exc:
- self._report(f"[WARNING] Could not unload previously loaded models at "
- f"{self.api_url}: {exc}")
- return
- unloaded = [entry for entry in (payload.get("unloaded") or [])
- if isinstance(entry, str)]
- if unloaded:
- self._report(f"[OK] Unloaded {len(unloaded)} model(s) from server memory: "
- f"{', '.join(unloaded)}")
- else:
- logger.debug("No loaded audio.cpp models to unload at %s", self.api_url)
-
- def _get_json(self, path: str, timeout: int = 10) -> Dict[str, Any]:
- """GET a JSON document from the server."""
- url = f"{self.api_url}{path}"
- try:
- with urllib.request.urlopen(url, timeout=timeout) as response:
- return json.loads(response.read().decode("utf-8"))
- except urllib.error.HTTPError as exc:
- detail = ""
- try:
- detail = exc.read().decode("utf-8", errors="replace")[:200]
- except Exception:
- pass
- raise RuntimeError(
- f"audio.cpp server returned HTTP {exc.code} for {path}: {detail}") from exc
- except urllib.error.URLError as exc:
- raise RuntimeError(f"audio.cpp request failed for {path}: {exc.reason}") from exc
-
- def _check_health(self) -> None:
- """Verify the server is reachable and reports healthy."""
- try:
- payload = self._get_json("/health")
- except Exception as exc:
- raise RuntimeError(
- f"audio.cpp server not reachable at {self.api_url}: {exc}. "
- "Start audiocpp_server first (see the 'audio.cpp backend' "
- "section of the README)."
- ) from exc
- if payload.get("status") != "ok":
- raise RuntimeError(
- f"The audio.cpp server at {self.api_url} reports status "
- f"{payload.get('status')!r} instead of 'ok'")
-
- def _list_models(self) -> List[Dict[str, str]]:
- """Fetch the (id, family, task) triples reported by the server."""
- try:
- payload = self._get_json("/v1/models")
- except Exception as exc:
- raise RuntimeError(
- f"The audio.cpp server at {self.api_url} did not answer "
- f"/v1/models: {exc}") from exc
- entries = payload.get("data") or []
- models: List[Dict[str, str]] = []
- for entry in entries:
- if isinstance(entry, dict) and entry.get("id"):
- models.append({
- "id": entry["id"],
- "family": entry.get("family") or "",
- "task": entry.get("task") or "",
- })
- return models
-
- def _auto_pick_model_id(self, models: List[Dict[str, str]]) -> None:
- """Resolve an empty model id when the server hosts exactly one entry.
-
- Multi-model servers generated with several lazily-loaded entries can
- be used without editing app/converter/config.py: leave AUDIOCPP_MODEL_ID
- (and ``--model``) unset, and the single hosted entry is chosen
- automatically. With more than one entry an explicit choice is required
- (via ``--model`` or AUDIOCPP_MODEL_ID), since guessing would risk
- synthesizing a whole book with the wrong family.
- """
- if self.model_id:
- return
- if len(models) == 1:
- self.model_id = models[0]["id"]
- logger.info(
- "AUDIOCPP_MODEL_ID is unset; using the only server entry '%s'",
- self.model_id)
- else:
- logger.debug(
- "AUDIOCPP_MODEL_ID is unset and the server hosts %d entries; "
- "an explicit --model or config id is required",
- len(models))
-
- def _require_model_id(self, models: List[Dict[str, str]]) -> None:
- """Verify the model id chosen for this run exists on the server.
-
- Speaker mode needs AUDIOCPP_MODEL_ID (the CustomVoice entry).
- Preset mode validates whichever id _select_model resolved, so a
- server hosting only a cloning model works for --voice. The default
- error distinguishes the two so the fix is obvious.
- """
- model_ids = [model["id"] for model in models]
- if self.model_id and self.model_id in model_ids:
- return
- configured = ", ".join(model_ids) or "none"
- if not self.model_id:
- raise RuntimeError(
- f"The audio.cpp server at {self.api_url} hosts {len(model_ids)} "
- f"model entries ({configured}); audiobook.py needs to know which "
- "one to use. Pass --model <id> when converting, or set "
- "AUDIOCPP_MODEL_ID in app/converter/config.py to one of them "
- "(see README)."
- )
- if self.preset_mode:
- raise RuntimeError(
- f"The audio.cpp server at {self.api_url} has no model id "
- f"'{self.model_id}' or clone model id "
- f"'{config.AUDIOCPP_CLONE_MODEL_ID}' (configured: {configured}). "
- "Add a TTS model entry for the family you want to the server "
- "config and match AUDIOCPP_MODEL_ID / AUDIOCPP_CLONE_MODEL_ID "
- "in app/converter/config.py to its id, or select it per run with "
- "--model (see README)."
- )
- raise RuntimeError(
- f"The audio.cpp server at {self.api_url} has no model id "
- f"'{self.model_id}' (configured: {configured}). Select the "
- "Qwen3-TTS CustomVoice entry for built-in speakers, or rerun "
- "with --voice NAME matching a voice_preset or voice_dir entry "
- "on any TTS model (see README)."
- )
-
- def _select_model(self, models: List[Dict[str, str]]) -> None:
- """Pick the model for preset (cloning) requests.
-
- Defaults to the primary model id. When AUDIOCPP_CLONE_MODEL_ID is
- configured and present on the server, preset requests are routed
- to it instead, so one server can host the CustomVoice model for
- speaker mode and the Base model for cloning (Qwen3-TTS setups).
- A clone id that names a model of a different family is ignored
- with a warning, since preset requests must synthesize with the
- family the run is configured for.
- """
- clone_model_id = config.AUDIOCPP_CLONE_MODEL_ID
- if not clone_model_id or clone_model_id == self.model_id:
- return
- families = {model["id"]: model["family"] for model in models}
- if clone_model_id not in families:
- # A qwen3_tts primary without its clone entry silently degrades
- # (presets are ignored on the CustomVoice model), so that case
- # keeps the warning; single-model servers of other families are
- # the normal configuration and only get a debug note.
- primary_family = families.get(self.model_id) or ""
- if primary_family == AUDIOCPP_FAMILY_QWEN3_TTS:
- logger.warning(
- "AUDIOCPP_CLONE_MODEL_ID %r is not configured on the audio.cpp "
- "server; preset requests use '%s' instead",
- clone_model_id, self.model_id)
- else:
- logger.debug(
- "AUDIOCPP_CLONE_MODEL_ID %r is not configured on the audio.cpp "
- "server; preset requests use '%s' instead",
- clone_model_id, self.model_id)
- return
- primary_family = families.get(self.model_id)
- clone_family = families[clone_model_id]
- if primary_family and clone_family and primary_family != clone_family:
- logger.warning(
- "AUDIOCPP_CLONE_MODEL_ID %r hosts family %r, but "
- "AUDIOCPP_MODEL_ID %r hosts %r; preset requests stay on "
- "'%s'. Point both ids at the same model entry in "
- "app/converter/config.py (single-model servers use the same id "
- "for both)",
- clone_model_id, clone_family, self.model_id, primary_family,
- self.model_id)
- return
- self.model_id = clone_model_id
-
- def _resolve_family(self, models: List[Dict[str, str]]) -> None:
- """Resolve the selected model's family and its request profile.
-
- The family comes from GET /v1/models; a missing family is an unknown
- family that falls through to the generic (clone-only) profile rather
- than guessing a specific one — audiocpp_server always reports family
- for entries its server.json describes.
- """
- entry = next(
- (model for model in models if model["id"] == self.model_id), None)
- family = (entry["family"] if entry is not None else "") or ""
- self.family = family
- self.profile = AUDIOCPP_FAMILY_PROFILES.get(
- family, AUDIOCPP_DEFAULT_FAMILY_PROFILE)
- if not family:
- logger.debug("Model '%s' reported no family; using the generic "
- "profile", self.model_id)
- elif family not in AUDIOCPP_FAMILY_PROFILES:
- logger.info(
- "audio.cpp family '%s' has no dedicated profile; using the "
- "generic profile (voice cloning via --voice, model-detected "
- "language)", family)
-
- def _resolve_task(self, models: List[Dict[str, str]]) -> None:
- """Resolve the selected model's task (tts, clon, vdes, ...) and set
- design mode for voice design entries.
-
- The task comes from GET /v1/models and is fixed per server entry by
- its server.json config (a VoiceDesign model must be hosted with
- "task": "vdes"). Servers that predate the task field hosted plain
- TTS models, so a missing task is treated as tts.
- """
- entry = next(
- (model for model in models if model["id"] == self.model_id), None)
- task = (entry["task"] if entry is not None else "") or ""
- if not task:
- task = AUDIOCPP_TASK_TTS
- logger.debug("Model '%s' reported no task; assuming tts",
- self.model_id)
- self.task = task
- self.design_mode = task == AUDIOCPP_TASK_VDES
-
- def _check_voice(self) -> None:
- """Verify the requested voice is available on the server.
-
- A voice name that matches no server preset or voice-library wav
- would be passed through to the model as a cached voice id; on the
- Base (cloning) model that is silently ignored and plain TTS audio
- comes back, so preset names are validated up front. When the
- voices endpoint cannot be queried, validation is skipped with a
- warning rather than blocking the run.
- """
- query = urllib.parse.urlencode({"model": self.model_id})
- try:
- payload = self._get_json(f"/v1/audio/voices?{query}")
- except Exception as exc:
- logger.warning("Could not list server voices; skipping voice "
- "validation: %s", exc)
- return
- voices = payload.get("voices") or []
- if self.voice not in voices:
- available = ", ".join(str(v) for v in voices) or "none"
- raise RuntimeError(
- f"Voice '{self.voice}' is not available on the audio.cpp server "
- f"(available: {available}). Configure it as a voice_preset or "
- "voice_dir entry in the server config, or pass a listed name "
- "with --voice (see README)."
- )
-
- # ------------------------------------------------------------------
- # HTTP requests
- # ------------------------------------------------------------------
-
- def _request_wav(self, text: str) -> bytes:
- """POST one sub-chunk and return the raw WAV bytes."""
- url = f"{self.api_url}/v1/audio/speech"
- payload: Dict[str, Any] = {
- "model": self.model_id,
- "input": text,
- }
- # Design models take no voice field (the voice comes from the
- # instruction); instruction-voice runs on families without built-in
- # speakers omit it too, since no speaker or preset was requested.
- if not self.design_mode and not self.instruction_voice:
- payload["voice"] = self.voice
- if self.profile.language_style == AUDIOCPP_LANG_DISPLAY:
- payload["language"] = self.language
- elif self.profile.language_style == AUDIOCPP_LANG_ISO:
- iso_code = LANGUAGE_ISO_CODES.get(self.language)
- if iso_code:
- payload["language"] = iso_code
- else:
- # "Auto": no code to send, so let the server pick its default.
- logger.debug("%s: no language code for %r; omitted from request",
- self.family, self.language)
- if self._seed >= 0:
- # audio.cpp has no negative "randomize" seed; a negative seed
- # means "let the server randomize", so the field is omitted.
- payload["seed"] = self._seed
- if self.instructions:
- # Explicit voice-design or style instruction (required for task
- # "vdes" entries; a Ctrl/style control on families that read it).
- payload["instructions"] = self.instructions
- elif not self.preset_mode and config.INSTRUCT \
- and self.profile.sends_instructions:
- # Style instruction for the Qwen3-TTS CustomVoice speakers;
- # ignored by the Base (cloning) model and other families.
- payload["instructions"] = config.INSTRUCT
- if self.request_options:
- # Generic per-model controls (--option KEY=VALUE): forwarded
- # verbatim; the model ignores keys it does not know.
- payload["options"] = dict(self.request_options)
- request = urllib.request.Request(
- url, data=json.dumps(payload).encode("utf-8"),
- headers={"Content-Type": "application/json"}, method="POST")
- timeout = config.API_TIMEOUT
- try:
- with urllib.request.urlopen(request, timeout=timeout) as response:
- wav = response.read()
- except urllib.error.HTTPError as exc:
- detail = ""
- try:
- detail = exc.read().decode("utf-8", errors="replace")[:200]
- except Exception:
- pass
- raise RuntimeError(f"audio.cpp server returned HTTP {exc.code}: {detail}") from exc
- except urllib.error.URLError as exc:
- raise RuntimeError(f"audio.cpp request failed: {exc.reason}") from exc
- if len(wav) < 12 or wav[:4] != b"RIFF" or wav[8:12] != b"WAVE":
- raise RuntimeError("audio.cpp server returned audio that is not a WAV file")
- return wav
-
- # ------------------------------------------------------------------
- # Chunk generation
- # ------------------------------------------------------------------
-
- def generate_chunk(self, text: str, chunk_num: int) -> Optional[str]:
- """Generate one audio chunk; returns its path in the chunks folder.
-
- The text is split into sub-requests of at most ``config.CHUNK_SIZE``
- words each; each sub-request returns a complete WAV file and the
- parts are concatenated into one chunk file.
- """
- try:
- sub_texts = split_into_chunks(text, max_words=config.CHUNK_SIZE)
- if not sub_texts:
- raise RuntimeError("No text to synthesize")
-
- output_path: Optional[Path] = None
- with tempfile.TemporaryDirectory(prefix="tts_parts_") as parts_dir, \
- self._chunk_heartbeat(chunk_num):
- part_paths = []
- for sub_num, sub_text in enumerate(sub_texts, 1):
- wav = self._request_wav(sub_text)
- destination = Path(parts_dir) / f"part_{sub_num:02d}.wav"
- destination.write_bytes(wav)
- part_paths.append(destination)
- if len(part_paths) == 1:
- output_path = self._chunk_path(chunk_num, ".wav")
- shutil.copy2(part_paths[0], output_path)
- else:
- output_path = self._chunk_path(chunk_num, ".wav")
- concat_audio_files(part_paths, output_path)
-
- logger.debug("Chunk %d generated successfully (%d sub-request(s))",
- chunk_num, len(sub_texts))
- return str(output_path)
-
- except ConversionCancelled:
- raise
- except Exception as exc:
- logger.error("audio.cpp chunk processing failed for chunk %d: %s",
- chunk_num, exc)
- return None