aboutsummaryrefslogtreecommitdiff
path: root/backends
diff options
context:
space:
mode:
Diffstat (limited to 'backends')
-rw-r--r--backends/__init__.py146
-rwxr-xr-xbackends/audiocpp.py1727
-rw-r--r--backends/common.py265
-rw-r--r--backends/envs.py185
-rwxr-xr-xbackends/faster.py416
-rw-r--r--backends/qwen.py272
-rw-r--r--backends/servers.py244
7 files changed, 0 insertions, 3255 deletions
diff --git a/backends/__init__.py b/backends/__init__.py
deleted file mode 100644
index 6a23ab7..0000000
--- a/backends/__init__.py
+++ /dev/null
@@ -1,146 +0,0 @@
-"""Registry of the TTS backends the audiobook generator can talk to.
-
-Each backend (audio.cpp, qwen, faster) lives in its own module and owns
-its setup wizard, its status detection, and the launch command it prints
-once configured. This package aggregates them into a single registry so
-``audiobook.py``'s TUI hub and future tools can iterate backends without
-hardcoding their names: ``backends.detect_all()`` reports which are set
-up (and whether their server is currently running), and the registry
-drives the hub's setup/configure menus.
-
-The registry is built lazily on the first call to ``get``/``detect_all``/
-``detect`` (not at package import time), because the backend modules pull
-in ``converter.tts`` and its third-party dependencies, which are only
-available inside the managed venv that ``audiobook.py`` bootstraps before
-importing them. ``backends.envs`` is imported during that bootstrap, so
-importing this package must stay cheap and dependency-free.
-
-Adding a backend: create ``backends/<name>.py`` exposing
-``detect() -> BackendStatus``, ``run_tui() -> int`` and
-``configure_actions: list[ConfigureAction]``, then append a ``BackendInfo`` in
-``_build_registry`` below. ``audiobook.py`` and the hub pick it up
-automatically.
-"""
-
-import shlex
-from dataclasses import dataclass, field
-from typing import Callable, List, Optional
-
-
-@dataclass
-class ServerSpec:
- """One launchable server process for a backend.
-
- A backend may expose more than one server (qwen runs CustomVoice and Base
- on separate ports). ARGV is the exact command line the hub spawns (using
- the managed venv's absolute binaries, so no shell activation is needed);
- URL is the endpoint ``common.server_running`` probes to decide readiness.
- """
- name: str
- url: str
- argv: List[str]
-
-
-@dataclass
-class BackendStatus:
- """How far a backend is set up, plus the command to start it.
-
- INSTALLED means the backend itself is present (a cloned + built
- checkout, or a pip package). CONFIGURED means the supporting files are
- in place (a server.json / voices.json and a converter/config.py that
- points at the right port). RUNNING means an external server is
- currently accepting connections on the configured port (probed by
- ``backends.common.server_running``). DETAILS are short status lines for
- the hub. LAUNCH_HINT is the human-readable command(s) the user runs to
- start the server, derived from SERVERS by ``format_launch_hint``.
- SERVERS is the machine-usable list of server processes the hub can
- start/stop (empty when the backend is not yet configured).
- """
- key: str
- label: str
- installed: bool
- configured: bool
- running: bool = False
- details: List[str] = field(default_factory=list)
- launch_hint: str = ""
- servers: List[ServerSpec] = field(default_factory=list)
-
- @property
- def ready(self) -> bool:
- """True when the backend is installed and configured for use."""
- return self.installed and self.configured
-
-
-def format_launch_hint(servers: List[ServerSpec]) -> str:
- """Join a backend's server argvs into a copy-pasteable launch hint."""
- return " ; ".join(shlex.join(s.argv) for s in servers)
-
-
-@dataclass
-class ConfigureAction:
- """A per-backend "configure" menu entry (e.g. "New server.json")."""
- label: str
- run: Callable[[], int]
-
-
-@dataclass
-class BackendInfo:
- """One registry entry: identity, detector, setup wizard, configure menu."""
- key: str
- label: str
- detect: Callable[[], BackendStatus]
- setup_tui: Callable[[], int]
- configure_actions: List[ConfigureAction] = field(default_factory=list)
-
-
-REGISTRY: List[BackendInfo] = []
-_BY_KEY: dict = {}
-
-
-def _build_registry() -> None:
- """Import the backend modules and wire up REGISTRY (once)."""
- if REGISTRY:
- return
- from . import audiocpp, faster, qwen
-
- REGISTRY.append(BackendInfo(
- key="audiocpp",
- label="audio.cpp",
- detect=audiocpp.detect,
- setup_tui=audiocpp.run_tui,
- configure_actions=audiocpp.configure_actions,
- ))
- REGISTRY.append(BackendInfo(
- key="qwen",
- label="qwen-tts",
- detect=qwen.detect,
- setup_tui=qwen.run_tui,
- configure_actions=qwen.configure_actions,
- ))
- REGISTRY.append(BackendInfo(
- key="faster",
- label="faster-qwen3-tts",
- detect=faster.detect,
- setup_tui=faster.run_tui,
- configure_actions=faster.configure_actions,
- ))
- for info in REGISTRY:
- _BY_KEY[info.key] = info
-
-
-def get(key: str) -> Optional[BackendInfo]:
- """Return the registry entry for KEY, or None."""
- _build_registry()
- return _BY_KEY.get(key)
-
-
-def detect_all() -> List[BackendStatus]:
- """Detect every registered backend's status, in registry order."""
- _build_registry()
- return [info.detect() for info in REGISTRY]
-
-
-def detect(key: str) -> Optional[BackendStatus]:
- """Detect a single backend by key."""
- info = get(key)
- return info.detect() if info is not None else None
diff --git a/backends/audiocpp.py b/backends/audiocpp.py
deleted file mode 100755
index 486017f..0000000
--- a/backends/audiocpp.py
+++ /dev/null
@@ -1,1727 +0,0 @@
-#!/usr/bin/env python3
-"""Set up the audio.cpp TTS backend for the audiobook generator.
-
-This does the whole audio.cpp setup end-to-end as a full-screen DOS-style
-TUI: locate or clone an audio.cpp checkout into ``./audio.cpp``, optionally
-build ``audiocpp_server``, pick model families/packages from the checkout's
-``model_specs`` catalog, transcribe reference .wav voices, write
-``server.json``, sync ``converter/config.py``, download the models, and
-print the exact command to start the server. It is driven by
-``audiobook.py``'s TUI hub (``backends.REGISTRY``) but can also be run
-directly for scripting — every value has a flag, and a non-interactive run
-with all flags supplied never opens the TUI.
-
-The converter is family-agnostic (it detects the family of the selected
-entry from ``GET /v1/models`` at startup), so any TTS family listed in the
-catalog works without further changes.
-
-Usage:
- python -m backends.audiocpp [--wavs WAV_DIR] [--output PATH]
- [--audiocpp-dir PATH] [--clone] [--families FAM1,FAM2]
- [--all-packages] [--host HOST] [--port PORT]
- [--build-backend {cuda,vulkan,hip,cpu}] [--backend {cuda,vulkan,hip,cpu}]
- [--lazy-load] [--whisper-model NAME] [--force]
- [--download] [--no-sync-port] [--no-sync-model-ids]
-
-With no flags and a terminal, the TUI wizard runs. Without a terminal
-(or with all flags supplied), it runs non-interactively from the flags;
-any missing required value is a hard error with a remediation hint.
-"""
-
-import argparse
-import json
-import os
-import re
-import subprocess
-import sys
-import urllib.parse
-from pathlib import Path
-from typing import Callable, Dict, List, Optional, Set, Tuple
-
-# Allow running directly (python backends/audiocpp.py) from any cwd.
-sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
-
-from backends import (
- BackendStatus,
- ConfigureAction,
- ServerSpec,
- common,
- format_launch_hint,
-)
-from backends.common import (
- CONFIG_PATH,
- PROMPT_TEXT_FILENAME,
- TTS_ROOT,
- VOICES_DIR,
- detect_wav_dir,
- find_wav_files,
- normalize_dir_arg,
- read_prompt_text,
- resolve_wav_dir_arg,
- write_prompt_text,
-)
-from backends.common import (
- wav_dir_info as _wav_dir_info,
-)
-from backends.common import (
- wav_dir_preview as _wav_dir_preview,
-)
-from converter import config
-from converter.tts import transcribe_reference_audio, whisper_backend_available
-from ui import tui
-
-DEFAULT_HOST = "127.0.0.1"
-FALLBACK_PORT = 8080
-
-BACKENDS = ("cuda", "vulkan", "hip", "cpu")
-
-TASK_TTS = "tts"
-TASK_VDES = "vdes"
-
-# audio.cpp is cloned into a sibling directory of the audiobook generator.
-AUDIOCPP_DIR_NAME = "audio.cpp"
-AUDIOCPP_GIT_URL = "https://github.com/0xShug0/audio.cpp"
-
-# Sentinel returned by tui.confirm (via its cancel_value) when the user
-# presses Esc on an overwrite prompt to go back to the checkout browser
-# instead of aborting the wizard.
-_GO_BACK = object()
-
-
-class _GoBack(Exception):
- """Raised inside the TUI wizard to fall back to the previous screen group.
-
- Every wizard widget is passed ``back_value=_GO_BACK`` so Esc returns the
- sentinel instead of aborting; pickers and confirmations that call into
- callbacks (task/id pickers, the transcription plan, the download prompt)
- convert that sentinel into this exception so the enclosing step can catch
- it and step back. Only the first screen (the checkout browser) lets Esc
- abort the whole wizard.
- """
-
-# Package names that mark a voice-design model (hosted with task "vdes").
-DESIGN_PACKAGE_RE = re.compile(r"voice[\s_\-]?design", re.IGNORECASE)
-
-# Short, friendly default entry ids for selected families. Other families
-# derive an id from their family name (see default_model_id). All families
-# are listed equally, in alphabetical order.
-PREFERRED_IDS = {
- "qwen3_tts": "qwen",
- "higgs_audio_tts": "higgs",
- "voxcpm2": "voxcpm2",
- "index_tts2": "indextts2",
-}
-
-
-class _TuiError(Exception):
- """A fatal error raised from inside the TUI wizard.
-
- The message is reported to stderr after the terminal is restored; the
- process exits with code 2 (matching a parser error).
- """
-
-
-def _interactive() -> bool:
- """True when the TUI wizard can run (curses importable + tty)."""
- try:
- import curses # noqa: F401
- except ImportError:
- return False
- try:
- return sys.stdin.isatty() and sys.stdout.isatty()
- except (AttributeError, ValueError):
- return False
-
-
-def _resolve_audiocpp_root(directory: Path) -> Optional[Path]:
- """Return the audio.cpp checkout root for DIRECTORY, or None.
-
- Accepts either the checkout root itself (it must contain a
- ``model_specs`` directory) or the ``model_specs`` directory inside
- it (the parent is used), so the file browser cannot pick the wrong
- one of the two.
- """
- if (directory / "model_specs").is_dir():
- return directory
- if directory.name == "model_specs" and directory.is_dir():
- return directory.parent
- return None
-
-
-def _audiocpp_root_status(directory: Path) -> Tuple[str, str]:
- """TUI status describing the directory listed in the checkout browser."""
- if _resolve_audiocpp_root(directory) is not None:
- return ("model_specs/ found here", "ok")
- return ("No model_specs/ directory here", "warn")
-
-
-def _audiocpp_root_preview(directory: Path) -> Optional[Tuple[str, str]]:
- """TUI status for a highlighted subdirectory in the checkout browser."""
- if (directory / "model_specs").is_dir():
- return ("contains model_specs/", "ok")
- return None
-
-
-def _checkout_auto_select(entry: Path) -> Optional[Path]:
- """Auto-accept a highlighted checkout in the TUI browser.
-
- A subdirectory named ``audio.cpp`` that already contains a
- ``model_specs`` directory is the audio.cpp checkout root, so it is
- accepted immediately on Enter/Right (as if ``[ Use this directory ]``
- had been pressed) instead of being descended into. Anything else
- returns None so the user keeps browsing. This is only consulted
- while auto-accepting is still enabled; after the user presses Esc to
- go back, the browser is restarted inside the previously accepted
- checkout and this callback is no longer passed, so a wrong guess can
- be corrected.
- """
- if entry.name == "audio.cpp" and (entry / "model_specs").is_dir():
- return entry
- return None
-
-
-# Backend display order, with short descriptions. The backend name is padded
-# so the descriptions' dashes line up in the menu.
-_BACKEND_DESCRIPTIONS = (
- ("cuda", "NVIDIA GPUs (fastest)"),
- ("vulkan", "cross-vendor GPU"),
- ("hip", "AMD GPUs"),
- ("cpu", "no GPU required"),
-)
-
-
-def _backend_options(detected: Optional[str] = None
- ) -> Tuple[List[Tuple[str, str]], int]:
- """Build the aligned backend menu options and the default index.
-
- The backend names are padded to a common width so the ``-`` dashes
- before the descriptions line up. When DETECTED matches one of the
- options, that option gets ``[auto-detected]`` appended and is the
- default (cursor/start) selection; otherwise the first option is the
- default as before. Returns (options, default_index).
- """
- width = max(len(name) for name, _ in _BACKEND_DESCRIPTIONS)
- options: List[Tuple[str, str]] = []
- default_index = 0
- for index, (name, desc) in enumerate(_BACKEND_DESCRIPTIONS):
- label = f"{name.ljust(width)} - {desc}"
- if detected == name:
- label += " [auto-detected]"
- default_index = index
- options.append((label, name))
- return options, default_index
-
-
-def config_port() -> int:
- """Return the port of AUDIOCPP_API_URL in converter/config.py."""
- try:
- return urllib.parse.urlsplit(config.AUDIOCPP_API_URL).port or FALLBACK_PORT
- except ValueError:
- return FALLBACK_PORT
-
-
-def _url_with_port(url: str, port: int) -> str:
- parts = urllib.parse.urlsplit(url)
- host = parts.hostname or "127.0.0.1"
- return urllib.parse.urlunsplit(
- (parts.scheme or "http", f"{host}:{port}", parts.path, "", ""))
-
-
-def update_config_api_url_port(port: int, config_path: Optional[Path] = None) -> bool:
- """Rewrite the port inside AUDIOCPP_API_URL in converter/config.py.
-
- Only the quoted URL literal is replaced; surrounding lines and the
- trailing comment are preserved. Returns True when the file was changed.
- """
- path = Path(config_path) if config_path is not None else CONFIG_PATH
- try:
- text = path.read_text(encoding="utf-8")
- except OSError:
- return False
- match = re.search(r'(?m)^(\s*AUDIOCPP_API_URL\s*=\s*")([^"]*)(")', text)
- if not match:
- return False
- new_url = _url_with_port(match.group(2), port)
- if new_url == match.group(2):
- return False
- text = text[:match.start(2)] + new_url + text[match.end(2):]
- try:
- path.write_text(text, encoding="utf-8")
- except OSError:
- return False
- return True
-
-
-def update_config_model_ids(model_id: str,
- clone_model_id: Optional[str] = None,
- config_path: Optional[Path] = None) -> bool:
- """Rewrite AUDIOCPP_MODEL_ID (and AUDIOCPP_CLONE_MODEL_ID when given).
-
- Only the quoted id literals are replaced; surrounding lines and
- comments are preserved. Returns True when the file was changed.
- """
- path = Path(config_path) if config_path is not None else CONFIG_PATH
- try:
- text = path.read_text(encoding="utf-8")
- except OSError:
- return False
- updates: List[Tuple[str, str]] = [("AUDIOCPP_MODEL_ID", model_id)]
- if clone_model_id is not None:
- updates.append(("AUDIOCPP_CLONE_MODEL_ID", clone_model_id))
- changed = False
- for name, value in updates:
- match = re.search(r'(?m)^(\s*' + name + r'\s*=\s*")([^"]*)(")', text)
- if match and match.group(2) != value:
- text = text[:match.start(2)] + value + text[match.end(2):]
- changed = True
- if not changed:
- return False
- try:
- path.write_text(text, encoding="utf-8")
- except OSError:
- return False
- return True
-
-
-def default_model_id(family: str) -> str:
- """Derive a default server entry id from a family name."""
- if family in PREFERRED_IDS:
- return PREFERRED_IDS[family]
- name = family
- if name.endswith("_tts"):
- name = name[:-4]
- return name.replace("_", "") or family
-
-
-def detect_audiocpp_dir() -> Optional[Path]:
- """Best-effort location of a local audio.cpp checkout with model_specs.
-
- Checks the AUDIOCPP_DIR environment variable, then an ``audio.cpp``
- directory in or above the current working directory. Returns the path
- only when it contains a ``model_specs`` directory.
- """
- candidates: List[Path] = []
- env_dir = os.environ.get("AUDIOCPP_DIR")
- if env_dir:
- candidates.append(Path(os.path.expanduser(env_dir)))
- cwd = Path.cwd()
- candidates.append(cwd / "audio.cpp")
- candidates.append(cwd.parent / "audio.cpp")
- candidates.append(cwd.parent.parent / "audio.cpp")
- for candidate in candidates:
- try:
- resolved = candidate.resolve()
- except OSError:
- continue
- if (resolved / "model_specs").is_dir():
- return resolved
- return None
-
-
-# audio.cpp build directories are named ``<platform>-<backend>-<type>`` (e.g.
-# ``linux-cuda-release``, ``windows-vulkan-debug``, ``macos-metal-release``)
-# and the built server lands in ``<that>/bin/audiocpp_server``. The Metal
-# macOS backend is reported as "cpu" here since it is not a separate
-# --backend choice for audiocpp_server.
-_BACKEND_TOKEN_RE = re.compile(r"-(cuda|vulkan|hip|cpu|metal)(?:-|$)")
-
-
-def detect_backend(audiocpp_dir: Path) -> Optional[str]:
- """Best-effort detection of the backend audiocpp_server was built for.
-
- Scans ``audiocpp_dir/build/*`` for build directories that contain a
- built ``bin/audiocpp_server`` (``.exe`` allowed on Windows) and reads
- the backend token out of the directory name (``-cuda-``, ``-vulkan-``,
- ``-hip-`` or ``-cpu-``; ``-metal-`` is mapped to ``cpu``). Returns the
- backend only when exactly one distinct backend was built, so a checkout
- with builds for several backends does not silently pick one. Returns
- None when there is no ``build/`` directory, no built server, or more
- than one distinct backend.
- """
- build_root = audiocpp_dir / "build"
- if not build_root.is_dir():
- return None
- backends: Set[str] = set()
- try:
- build_dirs = sorted(build_root.iterdir(),
- key=lambda p: p.name.lower())
- except OSError:
- return None
- for build_dir in build_dirs:
- if not build_dir.is_dir():
- continue
- server = build_dir / "bin" / "audiocpp_server"
- if not server.exists():
- server_exe = build_dir / "bin" / "audiocpp_server.exe"
- if not server_exe.exists():
- continue
- match = _BACKEND_TOKEN_RE.search(build_dir.name.lower())
- if not match:
- continue
- token = match.group(1)
- backends.add("cpu" if token == "metal" else token)
- if len(backends) == 1:
- return next(iter(backends))
- return None
-
-
-def _default_package(packages: List[dict]) -> Optional[dict]:
- """Pick the default package from a list of packages.
-
- Prefers the package flagged ``default: true``, then the first GGUF
- package, then the first package overall. Returns None for an empty list.
- """
- if not packages:
- return None
- for package in packages:
- if package.get("default"):
- return package
- for package in packages:
- if package.get("format") == "gguf":
- return package
- return packages[0]
-
-
-def load_model_catalog(audiocpp_dir: Path) -> List[dict]:
- """Read model_specs/*.json and return the TTS-capable families.
-
- Each returned entry has: family, display_name, description, languages,
- clone_capable, packages (the full list from the spec), install_id
- (recommended package id), default_path (``models/<target_directory>``),
- and preferred_id. All families are treated equally and listed in
- alphabetical order by display name.
- """
- specs_dir = audiocpp_dir / "model_specs"
- if not specs_dir.is_dir():
- raise NotADirectoryError(
- f"{audiocpp_dir} has no model_specs/ directory; point "
- "--audiocpp-dir at an audio.cpp checkout")
- entries: List[dict] = []
- for spec_path in sorted(specs_dir.glob("*.json")):
- try:
- spec = json.loads(spec_path.read_text(encoding="utf-8"))
- except (OSError, ValueError):
- continue
- tasks = spec.get("tasks") or []
- if "tts" not in tasks and spec.get("category") != "tts":
- continue
- family = spec.get("family") or spec_path.stem
- packages = spec.get("packages") or []
- package = _default_package(packages)
- if package is None:
- # No installable package: skip (cannot be hosted from a path).
- continue
- target_directory = package.get("target_directory") or family
- languages = spec.get("languages") or []
- display_name = spec.get("display_name") or family
- description = spec.get("description") or ""
- entries.append({
- "family": family,
- "display_name": display_name,
- "description": description,
- "languages": languages,
- "tasks": list(tasks),
- "clone_capable": "clone" in tasks,
- "packages": packages,
- "install_id": package.get("id") or family,
- "default_path": f"models/{target_directory}",
- "preferred_id": default_model_id(family),
- })
-
- # All families are treated equally: alphabetical by display name.
- entries.sort(key=lambda entry: entry["display_name"].lower())
- return entries
-
-
-def is_design_package(package: dict) -> bool:
- """Return True when a package's name marks it a voice-design model.
-
- audio.cpp voice-design packages (whose id, display name, or target
- directory mentions "voice design") are the only packages that must be
- hosted with task "vdes"; their role is not in the schema, only in those
- strings, so it is detected from them.
- """
- text = " ".join(str(package.get(key, ""))
- for key in ("id", "display_name", "target_directory"))
- return bool(DESIGN_PACKAGE_RE.search(text))
-
-
-def package_dir_options(entry: dict) -> List[dict]:
- """Return one option per distinct target_directory of a family's packages.
-
- Each option is a dict with: target_directory, install_id (the recommended
- package id inside that directory), design (voice-design package flag), and
- recommended (whether it holds the family's default package). Precisions
- that share a directory (q8_0/bf16/...) collapse to a single option.
- """
- packages = entry.get("packages") or []
- default_pkg = _default_package(packages)
- default_dir = (default_pkg or {}).get("target_directory") or entry["family"]
- by_dir: Dict[str, List[dict]] = {}
- order: List[str] = []
- for package in packages:
- directory = package.get("target_directory") or entry["family"]
- if directory not in by_dir:
- by_dir[directory] = []
- order.append(directory)
- by_dir[directory].append(package)
- options: List[dict] = []
- for directory in order:
- package = _default_package(by_dir[directory])
- options.append({
- "target_directory": directory,
- "install_id": (package or {}).get("id") or directory,
- "design": is_design_package(package or {}),
- "recommended": directory == default_dir,
- })
- # Put the recommended package first for a friendlier checklist.
- options.sort(key=lambda opt: not opt["recommended"])
- return options
-
-
-def build_model_entry(family: str, model_id: str, model_path: str,
- task: str = TASK_TTS) -> dict:
- """Assemble one server.json model entry.
-
- ``task`` defaults to "tts"; voice design packages are hosted with
- "vdes" so the server runs its design session for speech requests
- (audiobook.py then requires --instructions with that entry).
- """
- return {
- "id": model_id,
- "family": family,
- "path": model_path,
- "task": task,
- "mode": "offline",
- }
-
-
-def build_server_config(host: str, port: int, backend: str, lazy_load: bool,
- model_entries: List[dict],
- voice_dir: Optional[str] = None) -> dict:
- """Assemble the server.json document.
-
- ``voice_dir`` is a server-level cloning voice library; when set, every
- hosted clone-capable family can use its voices with ``--voice``.
- """
- config_doc = {
- "host": host,
- "port": port,
- "backend": backend,
- "lazy_load": lazy_load,
- "models": model_entries,
- }
- if voice_dir:
- config_doc["voice_dir"] = voice_dir
- return config_doc
-
-
-def transcribe_wav_dir(wav_files: list, whisper_model: str) -> Dict[str, str]:
- """Transcribe each wav file and return a mapping of stem -> transcript."""
- transcripts: Dict[str, str] = {}
- for wav_file in wav_files:
- name = wav_file.stem
- print(f"[INFO] Transcribing {wav_file.name} (voice '{name}')...")
- text = transcribe_reference_audio(str(wav_file), model_name=whisper_model)
- if text:
- print(f"[OK] {name}: {text}")
- else:
- print(f"[WARNING] No transcript for '{name}'; cloning works best "
- "with an accurate transcript — consider editing prompt_text "
- "by hand before starting the server")
- transcripts[name] = text or ""
- return transcripts
-
-
-def print_empty_transcript_warning(transcripts: Dict[str, str]) -> None:
- """Print a loud, final warning for voices whose transcript is empty."""
- empty = sorted(name for name, text in transcripts.items() if not text)
- if not empty:
- return
- bar = "=" * 70
- print()
- print(bar)
- print("[WARNING] MANUAL TRANSCRIPTION REQUIRED")
- print(bar)
- listing = " - " + "\n - ".join(empty) if len(empty) > 1 else f" - {empty[0]}"
- print(f"The following voice(s) have an EMPTY transcript in prompt_text:\n"
- f"{listing}")
- print("Those voices will NOT work until you add an accurate transcript.")
- print(f"Edit {PROMPT_TEXT_FILENAME} in your voice directory and fill in the "
- "text after '|' for each voice above.")
- print(bar)
-
-
-def _apply_port_sync(port: int, accepted: bool) -> None:
- """Write the port into converter/config.py, or report when declined."""
- if accepted:
- if not update_config_api_url_port(port):
- print(f"[WARNING] Could not update {CONFIG_PATH}; edit "
- "AUDIOCPP_API_URL by hand so audiobook.py uses the "
- "new port")
- else:
- print("[WARNING] Left AUDIOCPP_API_URL unchanged; audiobook.py "
- f"will still use port {config_port()}")
-
-
-def _decide_transcription(wav_files: list, existing: Dict[str, str],
- prompt_exists: bool, force: bool,
- confirm: Callable[[str, bool], bool]) -> dict:
- """Decide which voices to transcribe; CONFIRM asks the plan questions.
-
- Returns a plan dict: {"mode": "all"|"missing"|"keep", "missing":
- [...], "existing": {...}} — "existing" carries the prompt_text
- mapping read while deciding, so the caller can reuse it instead of
- reading the file again.
- """
- mode = "all"
- missing: List[Path] = []
- if prompt_exists and not force:
- missing = [wav for wav in wav_files
- if not existing.get(wav.stem, "").strip()]
- if not missing:
- if confirm("All voices already transcribed in prompt_text. "
- "Re-transcribe anyway?", False):
- mode = "all"
- else:
- mode = "keep"
- elif confirm("Existing transcription and new .wavs detected, "
- "only transcribe new voices?", True):
- mode = "missing"
- else:
- mode = "all"
- return {"mode": mode, "missing": missing, "existing": existing}
-
-
-def _transcribe(args: argparse.Namespace, include_clone: bool,
- plan: dict) -> Tuple[Dict[str, str], bool]:
- """Transcribe the wav directory into a stem -> transcript mapping.
-
- Returns the mapping and a flag indicating whether it should be written to
- prompt_text (False when an existing, complete prompt_text is kept as-is).
- PLAN is always pre-collected — by the TUI (via _decide_transcription and
- its confirm callbacks) or by _flag_plan for a non-interactive run — so no
- questions are asked here.
- """
- if not include_clone:
- print(f"[WARNING] Ignoring {args.input_dir}: no clone-capable family "
- "selected, so voice presets are not used")
- return {}, False
-
- wav_files = find_wav_files(args.input_dir)
- if not wav_files:
- print(f"[WARNING] No .wav files found in {args.input_dir}; writing the "
- "config without a voice_dir")
- return {}, False
-
- prompt_path = args.input_dir / PROMPT_TEXT_FILENAME
- existing = plan.get("existing") or {} if plan else {}
-
- if plan["mode"] == "keep":
- print(f"[INFO] Kept existing {prompt_path}; all voices were "
- "already transcribed, nothing new to transcribe")
- return existing, False
-
- if whisper_backend_available() is None:
- print("[WARNING] Neither faster_whisper nor whisper was found, so "
- "reference .wav files cannot be transcribed automatically and "
- "every transcript will be empty.")
- print(" Install whisper (or faster_whisper) in your "
- "audiobook environment to transcribe automatically; otherwise "
- "transcripts must be added by hand (see the warning at the end).")
-
- if plan["mode"] == "missing":
- new_transcripts = transcribe_wav_dir(plan["missing"], args.whisper_model)
- transcripts = dict(existing)
- transcripts.update(new_transcripts)
- else:
- transcripts = transcribe_wav_dir(wav_files, args.whisper_model)
- return transcripts, True
-
-
-def _flag_plan(wav_files: list, prompt_path: Path, force: bool) -> dict:
- """Build a transcription plan for a non-interactive (flag-only) run.
-
- With --force everything is re-transcribed; otherwise an existing
- prompt_text is reused and only voices with an empty transcript are
- re-transcribed, mirroring what the TUI confirms interactively.
- """
- if prompt_path.exists() and not force:
- existing = read_prompt_text(prompt_path)
- missing = [wav for wav in wav_files
- if not existing.get(wav.stem, "").strip()]
- if not missing:
- return {"mode": "keep", "missing": [], "existing": existing}
- return {"mode": "missing", "missing": missing, "existing": existing}
- return {"mode": "all", "missing": [], "existing": {}}
-
-
-def _offer_config_model_id_sync(model_id: str, accepted: Optional[bool]) -> None:
- """Point converter/config.py at a single hosted model entry.
-
- The converter requests the model id configured in AUDIOCPP_MODEL_ID,
- and single-model servers use the same id for the clone entry, so both
- ids are rewritten together. ACCEPTED is True/False (apply/skip the
- rewrite) or None when no single-entry sync applies (nothing to do).
- """
- if config.AUDIOCPP_MODEL_ID == model_id \
- and config.AUDIOCPP_CLONE_MODEL_ID == model_id:
- return
- if accepted is None:
- return
- if accepted:
- if not update_config_model_ids(model_id, model_id):
- print(f"[WARNING] Could not update {CONFIG_PATH}; edit "
- "AUDIOCPP_MODEL_ID and AUDIOCPP_CLONE_MODEL_ID by hand so "
- "audiobook.py uses this model")
- else:
- print("[WARNING] Left the model ids unchanged; audiobook.py will "
- f"still request model '{config.AUDIOCPP_MODEL_ID}'")
-
-
-def _build_entries(family_keys: List[str], chosen: Dict[str, List[dict]],
- catalog_by_family: Dict[str, dict],
- task_picker: Callable[[str], str],
- id_picker: Callable[[str, str, str], str]
- ) -> Tuple[List[dict], List[str], List[Tuple[str, str]],
- List[str], bool]:
- """Build server.json model entries from the selected families/packages.
-
- TASK_PICKER is called for each design package to choose vdes/tts;
- ID_PICKER resolves a duplicate server entry id. Returns (model_entries,
- entry_ids, install_guidance, design_entry_ids, include_clone).
- """
- model_entries: List[dict] = []
- entry_ids: List[str] = []
- install_guidance: List[Tuple[str, str]] = []
- design_entry_ids: List[str] = []
- include_clone = False
- for family in family_keys:
- entry = catalog_by_family[family]
- include_clone = include_clone or entry["clone_capable"]
- for opt in chosen[family]:
- task = task_picker(opt["install_id"]) if opt["design"] else TASK_TTS
- base_id = (f"{entry['preferred_id']}-design"
- if task == TASK_VDES else entry["preferred_id"])
- model_id = base_id
- if model_id in entry_ids:
- model_id = id_picker(entry["display_name"], opt["install_id"],
- f"{base_id}-2")
- entry_ids.append(model_id)
- model_entries.append(build_model_entry(
- family, model_id, f"models/{opt['target_directory']}",
- task=task))
- install_guidance.append((entry["display_name"], opt["install_id"]))
- if task == TASK_VDES:
- design_entry_ids.append(model_id)
- return (model_entries, entry_ids, install_guidance,
- design_entry_ids, include_clone)
-
-
-def _write_and_advise(audiocpp_dir: Path, wav_dir: Optional[Path],
- output_path: Path, model_entries: List[dict],
- install_guidance: List[Tuple[str, str]], host: str,
- port: int, backend: str, lazy_load: bool,
- transcripts: Dict[str, str], write_prompt: bool) -> None:
- """Console phase shared by both UI modes: write files, print summary.
-
- After a successful run the console output is the path of the written
- server.json. The model install commands (and optional automatic
- download) are handled separately by _install_models, called by both
- UI modes once the user has decided whether to download.
- """
- voice_dir: Optional[str] = None
- if transcripts:
- if write_prompt:
- prompt_path = wav_dir / PROMPT_TEXT_FILENAME
- write_prompt_text(wav_dir, transcripts)
- print(f"[OK] Wrote {prompt_path}")
- voice_dir = str(wav_dir.resolve())
-
- server_config = build_server_config(
- host=host, port=port, backend=backend, lazy_load=lazy_load,
- model_entries=model_entries, voice_dir=voice_dir)
-
- with output_path.open("w", encoding="utf-8") as handle:
- json.dump(server_config, handle, indent=2, ensure_ascii=False)
- handle.write("\n")
-
- count = len(model_entries)
- print(f"Wrote {output_path.resolve()} with {count} "
- f"{'entry' if count == 1 else 'entries'}.")
-
-
-def _install_models(audiocpp_dir: Path,
- install_guidance: List[Tuple[str, str]],
- download: bool) -> None:
- """Print and optionally run the model install commands.
-
- One ``python <manager> install <id>`` command per hosted model (de-duped
- by install id). When DOWNLOAD is True each command is run in the audio.cpp
- checkout via ``subprocess.run`` so the models are downloaded automatically;
- a failing install is reported as a warning and does not abort the remaining
- downloads. When DOWNLOAD is False (or the model manager is missing) the
- commands are only printed, copy-pasteable as before.
- """
- manager = audiocpp_dir / "tools" / "model_manager_v2.py"
- seen: Set[str] = set()
- install_ids: List[str] = []
- for _, install_id in install_guidance:
- if install_id not in seen:
- seen.add(install_id)
- install_ids.append(install_id)
-
- if download and not manager.is_file():
- print(f"[WARNING] {manager} not found; printing the install commands "
- "instead of running them")
- download = False
-
- for install_id in install_ids:
- command = f"python {manager} install {install_id}"
- if not download:
- print(command)
- continue
- print(f"[INFO] Downloading {install_id}...")
- try:
- result = subprocess.run(
- [sys.executable, str(manager), "install", install_id],
- cwd=str(audiocpp_dir))
- except OSError as exc:
- print(f"[WARNING] Could not run {command}: {exc}")
- continue
- if result.returncode != 0:
- print(f"[WARNING] install {install_id} exited with code "
- f"{result.returncode}; the model may need to be downloaded "
- "by hand")
-
-
-def _decide_download(audiocpp_dir: Path,
- confirm: Callable[[str, bool], bool]) -> bool:
- """Ask whether to download the selected models now.
-
- CONFIRM asks the yes/no question (ask_bool for the line prompts, a TUI
- confirm for the wizard). When the audio.cpp model manager is missing the
- prompt is skipped and False is returned, so the install commands are only
- printed rather than offered to run.
- """
- manager = audiocpp_dir / "tools" / "model_manager_v2.py"
- if not manager.is_file():
- return False
- return confirm(
- "Automatically download the selected models with model_manager_v2.py "
- "now?", False)
-
-
-def _build_tree_families(catalog: List[dict]) -> List[dict]:
- """Shape the catalog into the checkbox_tree widget's family list."""
- families: List[dict] = []
- for entry in catalog:
- capabilities = ["tts"]
- if "clone" in entry["tasks"]:
- capabilities.append("cloning")
- if "design" in entry["tasks"]:
- capabilities.append("design")
- name = entry["display_name"]
- if name != entry["family"]:
- name = f"{name} ({entry['family']})"
- options = []
- for opt in package_dir_options(entry):
- options.append({
- "key": opt["target_directory"],
- "label": opt["install_id"],
- "recommended": opt["recommended"],
- })
- families.append({
- "label": name,
- "detail": ", ".join(capabilities),
- "options": options,
- })
- return families
-
-
-def _wizard(stdscr, args: argparse.Namespace, parser: argparse.ArgumentParser
- ) -> Optional[dict]:
- """Run every TUI screen; return the collected settings, or None to abort.
-
- The wizard is a step state machine; each screen group is one step, and
- Esc anywhere but the first step falls back to the previous group (the
- widget returns the _GO_BACK sentinel, or a callback raises _GoBack). On
- the first screen (the audio.cpp checkout browser) Esc aborts the whole
- wizard as before.
- """
-
- def ask_confirm(question: str, default: bool) -> bool:
- result = tui.confirm(stdscr, question, default=default,
- cancel_value=_GO_BACK)
- if result is _GO_BACK:
- raise _GoBack()
- return result
-
- step = 0
- while True:
- if step == 0:
- # Checkout browser + the output path/overwrite confirmation. The
- # browser asks for the checkout root and finds model_specs/ inside
- # it (picking the model_specs directory itself works too — its
- # parent is used). A highlighted subdirectory named "audio.cpp"
- # that already contains model_specs/ is auto-accepted on
- # Enter/Right, skipping the "[ Use this directory ]" step.
- # Pressing Esc on an overwrite confirmation returns here instead
- # of aborting: the browser then restarts inside the previously
- # accepted checkout with auto-accept disabled, so a wrong guess
- # can be corrected. An explicit --audiocpp-dir flag has no
- # browser to return to, so Esc still aborts there. Esc on the
- # browser itself is the first step, so it aborts the wizard.
- auto_accept = True
- browser_start: Path = Path.cwd()
- force_browse = False
-
- def do_browse():
- return tui.browse_directory(
- stdscr, "Select your audio.cpp directory",
- validate=lambda p: None if _resolve_audiocpp_root(p)
- else "No model_specs/ directory here",
- info=_audiocpp_root_status,
- preview=_audiocpp_root_preview,
- help_lines=["The root folder of your audio.cpp "
- "checkout;",
- "it is the one that contains "
- "model_specs/"],
- start=browser_start,
- auto_select=_checkout_auto_select if auto_accept
- else None)
-
- while True:
- audiocpp_dir = args.audiocpp_dir
- if audiocpp_dir is None and not force_browse:
- audiocpp_dir = find_local_checkout()
- if force_browse:
- audiocpp_dir = None
- if audiocpp_dir is None:
- if force_browse:
- # Esc on an overwrite confirmation came back here: go
- # straight back into the browser inside the previously
- # accepted checkout (auto-accept disabled).
- audiocpp_dir = do_browse()
- else:
- # No checkout found anywhere: offer to clone one into
- # ./audio.cpp or browse for an existing checkout. Esc
- # on this first menu aborts the wizard.
- choice = tui.menu(
- stdscr, "No audio.cpp checkout found",
- [(f"Clone into ./{AUDIOCPP_DIR_NAME} "
- f"(from {AUDIOCPP_GIT_URL})", "clone"),
- ("Browse for an existing checkout", "browse")],
- help_lines=[
- "audio.cpp hosts the TTS model families "
- "this generator uses.",
- "Clone it into the project directory, or "
- "point at an existing checkout."])
- if choice == "clone":
- target = TTS_ROOT / AUDIOCPP_DIR_NAME
- with tui.suspend(stdscr):
- rc = common.git_clone(AUDIOCPP_GIT_URL,
- target)
- if rc != 0:
- raise _TuiError(
- f"git clone failed (exit {rc}). Clone "
- f"audio.cpp manually: git clone "
- f"{AUDIOCPP_GIT_URL} {target}")
- audiocpp_dir = target
- else:
- audiocpp_dir = do_browse()
- audiocpp_dir = Path(audiocpp_dir).resolve()
- if not audiocpp_dir.is_dir():
- raise _TuiError(f"audio.cpp checkout not found: "
- f"{audiocpp_dir}")
- root = _resolve_audiocpp_root(audiocpp_dir)
- if root is None:
- raise _TuiError(
- f"{audiocpp_dir} has no model_specs/ directory; "
- "select the root of your audio.cpp checkout")
- audiocpp_dir = root
- try:
- catalog = load_model_catalog(audiocpp_dir)
- except NotADirectoryError as exc:
- raise _TuiError(str(exc))
- if not catalog:
- raise _TuiError(f"No TTS model families found in "
- f"{audiocpp_dir}/model_specs; check the "
- "checkout is up to date")
- catalog_by_family = {entry["family"]: entry
- for entry in catalog}
-
- output_path = args.output if args.output is not None \
- else audiocpp_dir / "server.json"
- esc_back = args.audiocpp_dir is None
- went_back = False
- if not args.force and output_path.exists():
- decision = tui.confirm(
- stdscr, f"{output_path} already exists. Overwrite?",
- default=True,
- cancel_value=_GO_BACK if esc_back else None)
- if decision is _GO_BACK:
- went_back = True
- elif decision is False:
- if args.output is None:
- output_path = Path.cwd() / "server.json"
- if output_path.exists():
- decision = tui.confirm(
- stdscr,
- f"{output_path} already exists. "
- "Overwrite?",
- default=True,
- cancel_value=_GO_BACK if esc_back else None)
- if decision is _GO_BACK:
- went_back = True
- elif decision is False:
- return None
- else:
- return None
- if went_back:
- auto_accept = False
- browser_start = audiocpp_dir
- force_browse = True
- continue
- break
- detected_backend = detect_backend(audiocpp_dir)
- step = 1
- continue
-
- if step == 1:
- # Families and packages (flag or tree). Esc returns to the
- # checkout browser (step 0).
- chosen: Dict[str, List[dict]] = {}
- if args.families is not None:
- requested = [f.strip() for f in args.families.split(",")
- if f.strip()]
- unknown = [f for f in requested if f not in catalog_by_family]
- if unknown:
- raise _TuiError(
- f"Unknown family in --families: {', '.join(unknown)}. "
- f"Available: {', '.join(catalog_by_family)}")
- family_keys: List[str] = []
- for family in requested:
- if family not in family_keys:
- family_keys.append(family)
- chosen[family] = [opt for opt in package_dir_options(
- catalog_by_family[family]) if opt["recommended"]]
- else:
- tree_families = _build_tree_families(catalog)
- picked = tui.checkbox_tree(
- stdscr, "Select TTS model families to host",
- tree_families, expand_all=args.all_packages,
- back_value=_GO_BACK)
- if picked is _GO_BACK:
- step = 0
- continue
- family_keys = []
- for family_index, option_key in picked:
- family = catalog[family_index]["family"]
- if family not in chosen:
- chosen[family] = []
- family_keys.append(family)
- chosen[family].append(option_key)
- for family in list(chosen):
- keyed = {opt["target_directory"]: opt
- for opt in package_dir_options(
- catalog_by_family[family])}
- chosen[family] = [keyed[key] for key in chosen[family]]
- step = 2
- continue
-
- if step == 2:
- # Design task menus and duplicate-id renames. Esc anywhere here
- # falls back to the families tree (step 1).
- def task_picker(install_id: str) -> str:
- result = tui.menu(
- stdscr,
- f"How should the '{install_id}' package be hosted?",
- [
- ("design (vdes) - describe the voice with "
- "--instructions", TASK_VDES),
- ("tts - normal synthesis", TASK_TTS),
- ], default_index=0, back_value=_GO_BACK)
- if result is _GO_BACK:
- raise _GoBack()
- return result
-
- def id_picker(display_name: str, install_id: str,
- default: str) -> str:
- result = tui.line_edit(
- stdscr,
- f"Server model id for {display_name} package "
- f"'{install_id}'", default, back_value=_GO_BACK)
- if result is _GO_BACK:
- raise _GoBack()
- return result
-
- try:
- model_entries, entry_ids, install_guidance, \
- design_entry_ids, include_clone = _build_entries(
- family_keys, chosen, catalog_by_family,
- task_picker, id_picker)
- except _GoBack:
- step = 1
- continue
- step = 3
- continue
-
- if step == 3:
- # Server settings (host, port, port-sync, backend, lazy). Esc on
- # any of them falls back to the previous group (step 2).
- if args.host:
- host = args.host
- else:
- host = tui.line_edit(
- stdscr, "Bind host", DEFAULT_HOST,
- help_lines=["The IP address audiocpp will be hosted on",
- "127.0.0.1 (this machine) is probably "
- "correct"], back_value=_GO_BACK)
- if host is _GO_BACK:
- step = 2
- continue
- if args.port is not None:
- port = args.port
- else:
- port_text = tui.line_edit(
- stdscr, "Port", str(config_port()),
- validate=lambda s: None if (s.isdigit()
- and 1 <= int(s) <= 65535)
- else "Enter a port number between 1 and 65535",
- help_lines=["The port audiocpp will be hosted on"],
- back_value=_GO_BACK)
- if port_text is _GO_BACK:
- step = 2
- continue
- port = int(port_text)
- sync_port: Optional[bool] = None
- if port != config_port():
- sync_port = tui.confirm(
- stdscr, f"Update AUDIOCPP_API_URL in converter/config.py "
- f"to port {port} so audiobook.py talks to this server",
- default=True, cancel_value=_GO_BACK)
- if sync_port is _GO_BACK:
- step = 2
- continue
- if args.build_backend:
- backend = args.build_backend
- build = detected_backend is None
- elif args.backend:
- backend = args.backend
- build = False
- elif detected_backend is not None:
- # Already built: use the detected backend, no menu, no build.
- backend = detected_backend
- build = False
- else:
- backend_options, backend_default = _backend_options(None)
- backend = tui.menu(
- stdscr, "Which inference backend was audiocpp_server "
- "built for?", backend_options,
- default_index=backend_default, back_value=_GO_BACK)
- if backend is _GO_BACK:
- step = 2
- continue
- # Not built for any backend yet: offer to build it now. The
- # build itself runs in the console tail after the wizard.
- build = tui.confirm(
- stdscr, f"audiocpp_server is not built for {backend}. "
- f"Build it now (runs scripts/build_*)?",
- default=True, cancel_value=_GO_BACK)
- if build is _GO_BACK:
- step = 2
- continue
- default_lazy = len(model_entries) > 1
- if args.lazy_load:
- lazy_load = True
- else:
- lazy_load = tui.confirm(
- stdscr, "Load models lazily (on first use instead of at "
- "startup)", default=default_lazy, cancel_value=_GO_BACK)
- if lazy_load is _GO_BACK:
- step = 2
- continue
- step = 4
- continue
-
- if step == 4:
- # Wav directory (flag, browsed when cloning, else skipped). Esc
- # falls back to the server settings (step 3).
- if args.input_dir is not None:
- wav_dir = args.input_dir
- elif include_clone:
- wav_start = detect_wav_dir(audiocpp_dir, TTS_ROOT)
- wav_dir = tui.browse_directory(
- stdscr, "Select the directory with your .wav voices",
- info=_wav_dir_info, preview=_wav_dir_preview,
- start=wav_start if wav_start is not None else VOICES_DIR,
- back_value=_GO_BACK)
- if wav_dir is _GO_BACK:
- step = 3
- continue
- else:
- wav_dir = None
- step = 5
- continue
-
- if step == 5:
- # Transcription plan (questions only; transcription runs after).
- # Esc falls back to the wav browser (step 4).
- plan: Optional[dict] = None
- if include_clone and wav_dir is not None:
- wav_files = find_wav_files(wav_dir)
- if wav_files:
- prompt_path = wav_dir / PROMPT_TEXT_FILENAME
- existing = read_prompt_text(prompt_path) if (
- prompt_path.exists() and not args.force) else {}
- try:
- plan = _decide_transcription(
- wav_files, existing, prompt_path.exists(),
- args.force, ask_confirm)
- except _GoBack:
- step = 4
- continue
- step = 6
- continue
-
- if step == 6:
- # Single-model id sync decision. Esc falls back to the
- # transcription plan (step 5).
- sync_model_ids: Optional[bool] = None
- if len(entry_ids) == 1 and not (
- config.AUDIOCPP_MODEL_ID == entry_ids[0]
- and config.AUDIOCPP_CLONE_MODEL_ID == entry_ids[0]):
- sync_model_ids = tui.confirm(
- stdscr, "Update AUDIOCPP_MODEL_ID and "
- "AUDIOCPP_CLONE_MODEL_ID in converter/config.py to "
- f"'{entry_ids[0]}' so audiobook.py uses this model",
- default=True, cancel_value=_GO_BACK)
- if sync_model_ids is _GO_BACK:
- step = 5
- continue
- step = 8
- continue
-
- if step == 8:
- # Automatic model download (or print the install commands). Esc
- # falls back to the model-id sync (step 6).
- try:
- download = _decide_download(audiocpp_dir, ask_confirm)
- except _GoBack:
- step = 6
- continue
- return {
- "audiocpp_dir": audiocpp_dir,
- "catalog": catalog,
- "catalog_by_family": catalog_by_family,
- "output_path": output_path,
- "family_keys": family_keys,
- "chosen": chosen,
- "model_entries": model_entries,
- "entry_ids": entry_ids,
- "install_guidance": install_guidance,
- "design_entry_ids": design_entry_ids,
- "include_clone": include_clone,
- "host": host,
- "port": port,
- "backend": backend,
- "build": build,
- "lazy_load": lazy_load,
- "sync_port": sync_port,
- "sync_model_ids": sync_model_ids,
- "wav_dir": wav_dir,
- "plan": plan,
- "download": download,
- }
-
-
-def find_local_checkout() -> Optional[Path]:
- """Best-effort location of an audio.cpp checkout with model_specs.
-
- Checks the AUDIOCPP_DIR environment variable, then ``./audio.cpp`` inside
- the tts-audiobook-generator root, then an ``audio.cpp`` directory in or
- above the current working directory. Returns the path only when it
- contains a ``model_specs`` directory.
- """
- candidates: List[Path] = []
- env_dir = os.environ.get("AUDIOCPP_DIR")
- if env_dir:
- candidates.append(Path(os.path.expanduser(env_dir)))
- candidates.append(TTS_ROOT / AUDIOCPP_DIR_NAME)
- cwd = Path.cwd()
- candidates.append(cwd / AUDIOCPP_DIR_NAME)
- candidates.append(cwd.parent / AUDIOCPP_DIR_NAME)
- candidates.append(cwd.parent.parent / AUDIOCPP_DIR_NAME)
- for candidate in candidates:
- try:
- resolved = candidate.resolve()
- except OSError:
- continue
- if (resolved / "model_specs").is_dir():
- return resolved
- return None
-
-
-def find_audiocpp_server_bin(audiocpp_dir: Path) -> Optional[Path]:
- """Return the built audiocpp_server binary, or None when not built.
-
- Scans ``audiocpp_dir/build/*`` for a build directory containing
- ``bin/audiocpp_server`` (``.exe`` allowed on Windows). When several
- builds exist the first (alphabetical) is returned.
- """
- build_root = audiocpp_dir / "build"
- if not build_root.is_dir():
- return None
- try:
- build_dirs = sorted(build_root.iterdir(),
- key=lambda p: p.name.lower())
- except OSError:
- return None
- for build_dir in build_dirs:
- if not build_dir.is_dir():
- continue
- for name in ("audiocpp_server", "audiocpp_server.exe"):
- server = build_dir / "bin" / name
- if server.exists():
- return server
- return None
-
-
-def find_build_script(audiocpp_dir: Path) -> Optional[Path]:
- """Return the audio.cpp build helper script to run, or None.
-
- Prefers ``scripts/build_linux.sh``; otherwise the first
- ``scripts/build_*.sh`` it finds. (Windows ``.bat`` scripts are not run
- automatically — build manually there.)
- """
- scripts = audiocpp_dir / "scripts"
- if not scripts.is_dir():
- return None
- preferred = scripts / "build_linux.sh"
- if preferred.exists():
- return preferred
- try:
- candidates = sorted(scripts.glob("build_*.sh"),
- key=lambda p: p.name.lower())
- except OSError:
- return None
- return candidates[0] if candidates else None
-
-
-def build_audiocpp(audiocpp_dir: Path, backend: str) -> int:
- """Build audiocpp_server for BACKEND, streaming output to the console.
-
- Returns the build script's exit code (non-zero when the script is
- missing). Run from a console context (after the TUI wizard returns, or
- inside ``tui.suspend``).
- """
- script = find_build_script(audiocpp_dir)
- if script is None:
- print(f"[ERROR] No build script found in {audiocpp_dir}/scripts; "
- "build audiocpp_server manually (see the audio.cpp README)")
- return 1
- print(f"[INFO] Building audiocpp_server for {backend} "
- f"({script} --backend {backend} --target audiocpp_server)...")
- return common.run_console_subprocess(
- ["sh", str(script), "--backend", backend, "--target",
- "audiocpp_server"],
- cwd=audiocpp_dir)
-
-
-def _print_launch_hint(audiocpp_dir: Path, output_path: Path) -> None:
- """Print the exact command to start the server (or build guidance)."""
- binary = find_audiocpp_server_bin(audiocpp_dir)
- print()
- if binary is not None:
- print("Start the server with:")
- print(f" {binary} --config {output_path}")
- else:
- print("[INFO] audiocpp_server binary not found. Build it first, e.g.:")
- script = find_build_script(audiocpp_dir)
- if script is not None:
- print(f" sh {script} --backend <cuda|vulkan|hip|cpu> "
- "--target audiocpp_server")
- print(f" then run: ./build/<platform>-<backend>-release/bin/"
- f"audiocpp_server --config {output_path}")
-
-
-def _execute(settings: dict, args: argparse.Namespace) -> int:
- """Shared console tail: build, sync, transcribe, write, install, advise.
-
- Runs after the TUI wizard returns (or after _collect_from_flags for a
- non-interactive run): the terminal is plain, so subprocess output and
- transcription progress appear normally.
- """
- audiocpp_dir = settings["audiocpp_dir"]
-
- # Build audiocpp_server first (the longest step), when requested.
- if settings.get("build"):
- rc = build_audiocpp(audiocpp_dir, settings["backend"])
- if rc != 0:
- print(f"[WARNING] build exited with code {rc}; the server.json "
- "was still written — build audiocpp_server manually before "
- "starting it")
- else:
- print("[OK] build complete")
-
- # Port sync (applied now that the terminal is back).
- if settings["sync_port"] is True:
- _apply_port_sync(settings["port"], True)
- elif settings["sync_port"] is False:
- _apply_port_sync(settings["port"], False)
-
- # Transcription (console; the questions were already answered).
- args.input_dir = settings["wav_dir"]
- if settings["include_clone"] and args.input_dir is not None:
- transcripts, write_prompt = _transcribe(args, True, plan=settings["plan"])
- elif args.input_dir is not None:
- print(f"[WARNING] Ignoring {args.input_dir}: no clone-capable family "
- "selected, so voice presets are not used")
- transcripts, write_prompt = {}, False
- else:
- transcripts, write_prompt = {}, False
-
- _write_and_advise(
- audiocpp_dir, settings["wav_dir"], settings["output_path"],
- settings["model_entries"], settings["install_guidance"],
- settings["host"], settings["port"], settings["backend"],
- settings["lazy_load"], transcripts, write_prompt)
-
- if len(settings["entry_ids"]) == 1:
- _offer_config_model_id_sync(settings["entry_ids"][0],
- settings["sync_model_ids"])
- print_empty_transcript_warning(transcripts)
- _install_models(audiocpp_dir, settings["install_guidance"],
- settings["download"])
- _print_launch_hint(audiocpp_dir, settings["output_path"])
- return 0
-
-
-def run_tui(args: Optional[argparse.Namespace] = None,
- parser: Optional[argparse.ArgumentParser] = None) -> int:
- """Run the audio.cpp setup wizard end-to-end.
-
- With no ARGS (the hub's call) a default namespace is built so the full
- wizard runs. Called from ``main`` after argparse when the terminal is
- interactive. Returns the process exit code.
- """
- import curses
- if args is None:
- parser = build_parser()
- args = parser.parse_args([])
- if args.input_dir is not None and not args.input_dir.is_dir():
- print(f"[ERROR] --wavs not found: {args.input_dir}",
- file=sys.stderr)
- return 2
- try:
- settings = curses.wrapper(_wizard, args, parser)
- except _TuiError as exc:
- print(f"[ERROR] {exc}", file=sys.stderr)
- return 2
- except tui.WizardCancelled:
- print("\n[INFO] Cancelled; nothing was written")
- return 1
- try:
- curses.curs_set(1) # restore the text cursor hidden by the TUI
- except curses.error:
- pass
- if settings is None:
- print("[INFO] Aborted; existing server.json kept")
- return 1
- return _execute(settings, args)
-
-
-def _collect_from_flags(args: argparse.Namespace,
- parser: argparse.ArgumentParser) -> Optional[dict]:
- """Build the settings dict from flags for a non-interactive run.
-
- Every required value must come from a flag (there are no prompts in a
- non-interactive run); a missing one is a hard ``parser.error``. Returns
- the settings dict, or None when the user declined an overwrite (the
- default-location fallback then also exists).
- """
- # Checkout: --audiocpp-dir, else a local checkout, else --clone clones one.
- audiocpp_dir = args.audiocpp_dir
- if audiocpp_dir is None:
- audiocpp_dir = find_local_checkout()
- if audiocpp_dir is None and args.clone:
- target = TTS_ROOT / AUDIOCPP_DIR_NAME
- rc = common.git_clone(AUDIOCPP_GIT_URL, target)
- if rc != 0:
- parser.error(f"git clone failed (exit {rc}); clone audio.cpp "
- f"manually: git clone {AUDIOCPP_GIT_URL} {target}")
- audiocpp_dir = target
- if audiocpp_dir is None:
- parser.error(
- "An audio.cpp checkout is required. Pass --audiocpp-dir PATH, "
- "or --clone to clone ./audio.cpp, or run without flags for the "
- "TUI wizard.")
- audiocpp_dir = Path(audiocpp_dir).resolve()
- if not audiocpp_dir.is_dir():
- parser.error(f"audio.cpp checkout not found: {audiocpp_dir}")
- root = _resolve_audiocpp_root(audiocpp_dir)
- if root is None:
- parser.error(f"{audiocpp_dir} has no model_specs/ directory; point "
- "--audiocpp-dir at the root of an audio.cpp checkout")
- audiocpp_dir = root
- try:
- catalog = load_model_catalog(audiocpp_dir)
- except NotADirectoryError as exc:
- parser.error(str(exc))
- if not catalog:
- parser.error(
- f"No TTS model families found in {audiocpp_dir}/model_specs; "
- "check the checkout is up to date")
- catalog_by_family = {entry["family"]: entry for entry in catalog}
-
- # Families: required from --families in a non-interactive run.
- if args.families is None:
- parser.error("--families is required in a non-interactive run (or run "
- "without flags for the TUI wizard)")
- requested = [f.strip() for f in args.families.split(",") if f.strip()]
- unknown = [f for f in requested if f not in catalog_by_family]
- if unknown:
- parser.error(
- f"Unknown family in --families: {', '.join(unknown)}. "
- f"Available: {', '.join(catalog_by_family)}")
- family_keys: List[str] = []
- for fam in requested:
- if fam not in family_keys:
- family_keys.append(fam)
-
- chosen: Dict[str, List[dict]] = {}
- for family in family_keys:
- opts = package_dir_options(catalog_by_family[family])
- if args.all_packages:
- chosen[family] = opts
- else:
- chosen[family] = [opt for opt in opts if opt["recommended"]]
-
- # Non-interactive pickers: design packages default to vdes, dup ids get -2.
- def task_picker(install_id: str) -> str:
- return TASK_VDES
-
- def id_picker(display_name: str, install_id: str, default: str) -> str:
- return default
-
- model_entries, entry_ids, install_guidance, design_entry_ids, include_clone = \
- _build_entries(family_keys, chosen, catalog_by_family,
- task_picker, id_picker)
-
- # Server settings.
- host = args.host or DEFAULT_HOST
- detected_backend = detect_backend(audiocpp_dir)
- if args.build_backend:
- backend = args.build_backend
- build = detected_backend is None
- elif args.backend:
- backend = args.backend
- build = False
- elif detected_backend is not None:
- backend = detected_backend
- build = False
- else:
- backend = "cuda"
- build = False
- port = args.port if args.port is not None else config_port()
- lazy_load = args.lazy_load if args.lazy_load else (len(model_entries) > 1)
-
- # Output path / overwrite (decline falls back to cwd, then aborts).
- output_path = args.output if args.output is not None \
- else audiocpp_dir / "server.json"
- if output_path.exists() and not args.force:
- if args.output is None:
- output_path = Path.cwd() / "server.json"
- if output_path.exists() and not args.force:
- print("[INFO] Aborted; existing server.json kept")
- return None
- else:
- print("[INFO] Aborted; existing server.json kept")
- return None
-
- # Config sync decisions (auto-apply unless explicitly declined).
- sync_port: Optional[bool] = None
- if port != config_port():
- sync_port = not args.no_sync_port
- sync_model_ids: Optional[bool] = None
- if len(entry_ids) == 1 and not (
- config.AUDIOCPP_MODEL_ID == entry_ids[0]
- and config.AUDIOCPP_CLONE_MODEL_ID == entry_ids[0]):
- sync_model_ids = not args.no_sync_model_ids
-
- # Wav dir + transcription plan (defaults to the project's voices/ dir).
- wav_dir = args.input_dir if args.input_dir is not None else VOICES_DIR
- plan: Optional[dict] = None
- if include_clone and wav_dir is not None:
- wav_files = find_wav_files(wav_dir)
- if wav_files:
- prompt_path = wav_dir / PROMPT_TEXT_FILENAME
- plan = _flag_plan(wav_files, prompt_path, args.force)
-
- return {
- "audiocpp_dir": audiocpp_dir,
- "catalog": catalog,
- "catalog_by_family": catalog_by_family,
- "output_path": output_path,
- "family_keys": family_keys,
- "chosen": chosen,
- "model_entries": model_entries,
- "entry_ids": entry_ids,
- "install_guidance": install_guidance,
- "design_entry_ids": design_entry_ids,
- "include_clone": include_clone,
- "host": host,
- "port": port,
- "backend": backend,
- "build": build,
- "lazy_load": lazy_load,
- "sync_port": sync_port,
- "sync_model_ids": sync_model_ids,
- "wav_dir": wav_dir,
- "plan": plan,
- "download": args.download,
- }
-
-
-def build_parser() -> argparse.ArgumentParser:
- """The audio.cpp setup CLI (also used to build a default namespace)."""
- parser = argparse.ArgumentParser(
- description="Set up the audio.cpp TTS backend: clone/build, pick "
- "models, write server.json, and sync converter/config.py.")
- parser.add_argument("--wavs", type=resolve_wav_dir_arg, default=None,
- dest="input_dir", metavar="WAV_DIR",
- help="Directory with .wav reference files to publish as "
- "a server-level voice_dir cloning library "
- f"(default: {VOICES_DIR}; asked for when omitted "
- "in the TUI)")
- parser.add_argument("--output", type=Path, default=None,
- help="Output path for server.json (default: "
- "server.json inside the audio.cpp checkout; an "
- "existing file is overwritten only with --force "
- "or a TUI confirm)")
- parser.add_argument("--audiocpp-dir", type=normalize_dir_arg, default=None,
- help="Path to a local audio.cpp checkout containing a "
- "model_specs/ directory (default: detected from "
- "AUDIOCPP_DIR or ./audio.cpp; in the TUI you can "
- "clone one instead)")
- parser.add_argument("--clone", action="store_true",
- help="Non-interactive: clone audio.cpp into "
- "./audio.cpp when no checkout is found")
- parser.add_argument("--families", type=str, default=None,
- help="Comma-separated model families to host, as named "
- "in the audio.cpp catalog (e.g. "
- "qwen3_tts,higgs_audio_tts). Required in a "
- "non-interactive run; skips the family tree in "
- "the TUI")
- parser.add_argument("--all-packages", action="store_true",
- help="Host every installable package of each selected "
- "family (distinct target_directory) instead of "
- "only the recommended one. Voice-design packages "
- "are hosted with task 'vdes'")
- parser.add_argument("--host", type=str, default=None,
- help="Bind host for the server (default: 127.0.0.1)")
- parser.add_argument("--port", type=int, default=None,
- help="Port for the server (default: the port in "
- "AUDIOCPP_API_URL from converter/config.py)")
- parser.add_argument("--backend", choices=BACKENDS, default=None,
- help="Inference backend recorded in server.json "
- "(default: auto-detected from the checkout's "
- "build/ directory, else cuda)")
- parser.add_argument("--build-backend", choices=BACKENDS, default=None,
- help="Build audiocpp_server for this backend when it "
- "is not built yet, and use it in server.json")
- parser.add_argument("--lazy-load", action="store_true",
- help="Load models on first use instead of at startup "
- "(default: on when more than one model is hosted)")
- parser.add_argument("--whisper-model", type=str, default="base",
- help="Whisper model size for transcription "
- "(default: base)")
- parser.add_argument("--force", action="store_true",
- help="Overwrite the output file (and prompt_text) "
- "without prompting")
- parser.add_argument("--download", action="store_true",
- help="Run model_manager_v2.py install for each hosted "
- "model automatically (default: print the commands "
- "only)")
- parser.add_argument("--no-sync-port", action="store_true",
- help="Do not rewrite AUDIOCPP_API_URL in "
- "converter/config.py when --port differs")
- parser.add_argument("--no-sync-model-ids", action="store_true",
- help="Do not rewrite AUDIOCPP_MODEL_ID/"
- "AUDIOCPP_CLONE_MODEL_ID for a single-entry server")
- return parser
-
-
-def detect() -> BackendStatus:
- """Detect how far audio.cpp is set up, plus the command to start it."""
- checkout = find_local_checkout()
- # Probe the server first: it may be running externally even with no
- # local checkout, and the status table should show that.
- running = common.server_running(config.AUDIOCPP_API_URL)
- details: List[str] = []
- launch = ""
- if checkout is None:
- return BackendStatus("audiocpp", "audio.cpp", installed=False,
- configured=False, running=running,
- details=["not cloned — run setup to clone "
- "./audio.cpp"])
- details.append(f"checkout: {checkout}")
- binary = find_audiocpp_server_bin(checkout)
- built = binary is not None
- if built:
- details.append(f"built: {binary}")
- else:
- details.append("not built — run setup to build audiocpp_server")
- server_json = checkout / "server.json"
- configured = server_json.exists()
- servers: List[ServerSpec] = []
- if configured:
- details.append(f"config: {server_json}")
- if built:
- servers = [ServerSpec(
- "audiocpp", config.AUDIOCPP_API_URL,
- [str(binary), "--config", str(server_json)])]
- else:
- launch = (f"./build/<platform>-<backend>-release/bin/"
- f"audiocpp_server --config {server_json}")
- else:
- details.append("no server.json — run setup to configure models")
- if servers:
- launch = format_launch_hint(servers)
- return BackendStatus("audiocpp", "audio.cpp", installed=built,
- configured=configured, running=running,
- details=details, launch_hint=launch,
- servers=servers)
-
-
-configure_actions: List[ConfigureAction] = [
- ConfigureAction("Reconfigure audio.cpp (models, voices, server.json)",
- run_tui),
-]
-
-
-def main() -> int:
- parser = build_parser()
- args = parser.parse_args()
-
- if args.input_dir is not None and not args.input_dir.is_dir():
- parser.error(
- f"WAV directory not found: {args.input_dir}\n"
- f" (resolved from the current working directory: "
- f"{Path.cwd()})\n"
- " --wavs must be a directory containing the .wav "
- "reference files to use as voice cloning presets")
-
- if _interactive():
- return run_tui(args, parser)
-
- # Non-interactive (no terminal, or all flags supplied): flag-only path.
- settings = _collect_from_flags(args, parser)
- if settings is None:
- return 1
- return _execute(settings, args)
-
-
-if __name__ == "__main__":
- sys.exit(main())
diff --git a/backends/common.py b/backends/common.py
deleted file mode 100644
index 2529a8f..0000000
--- a/backends/common.py
+++ /dev/null
@@ -1,265 +0,0 @@
-"""Shared helpers for the backend setup wizards.
-
-Every TTS backend setup wizard (audio.cpp, qwen, faster) lives in its own
-module under ``backends``; this module holds the pieces more than one of
-them needs: .wav discovery, path normalization, and the regex edit that
-keeps ``converter/config.py`` in sync with the choices made in a wizard.
-It deliberately imports nothing from the other backend modules (or the
-TUI) so it can be reused without pulling curses into a non-interactive
-run.
-"""
-
-import os
-import re
-import urllib.parse
-from pathlib import Path
-from typing import Dict, List, Optional, Set, Tuple
-
-# The tts-audiobook-generator checkout root (where audiobook.py lives).
-# Backend checkouts are cloned into subdirectories of this root
-# (./audio.cpp, ./faster-qwen3-tts) so a single tree holds everything.
-TTS_ROOT = Path(__file__).resolve().parent.parent
-
-# The project's sample-voice directory: .wav files dropped here are offered
-# as the default source when a setup/configure wizard asks for a wav
-# directory (both the TUI browser start and the --wavs flag default).
-VOICES_DIR = TTS_ROOT / "voices"
-
-# converter/config.py — rewritten in place by update_config_value so the
-# converter picks up the host/port/voice a wizard configured.
-CONFIG_PATH = TTS_ROOT / "converter" / "config.py"
-
-# Output directory of tts-audiobook-generator; never offered as a .wav
-# source by detect_wav_dir.
-TTS_OUTPUT_DIR = "output"
-
-# The voice-transcript mapping file audio.cpp reads from its voice_dir.
-# (The faster backend uses voices.json instead; see backends.faster.)
-PROMPT_TEXT_FILENAME = "prompt_text"
-
-
-def normalize_dir_arg(value: str) -> Path:
- """Normalize a user-supplied path argument.
-
- Strips surrounding quotes (a common copy-paste artifact), expands a
- leading ``~``, and resolves the result to an absolute path so relative
- paths are always validated against the current working directory.
- """
- cleaned = value.strip()
- if len(cleaned) >= 2 and cleaned[0] == cleaned[-1] and cleaned[0] in "\"'":
- cleaned = cleaned[1:-1]
- return Path(os.path.expanduser(cleaned)).resolve()
-
-
-def resolve_wav_dir_arg(value: str) -> Path:
- """Normalize a user-supplied wav directory argument."""
- return normalize_dir_arg(value)
-
-
-def find_wav_files(input_dir: Path) -> List[Path]:
- """Return the .wav files in INPUT_DIR, sorted alphabetically by name."""
- return sorted(
- (path for path in input_dir.iterdir()
- if path.is_file() and path.suffix.lower() == ".wav"),
- key=lambda path: path.name.lower(),
- )
-
-
-def count_wavs(directory: Path) -> int:
- """Count the .wav files in DIRECTORY (0 when it cannot be read)."""
- try:
- return sum(1 for path in directory.iterdir()
- if path.is_file() and path.suffix.lower() == ".wav")
- except OSError:
- return 0
-
-
-def detect_wav_dir(audiocpp_dir: Path, tts_root: Path) -> Optional[Path]:
- """Find a unique directory that directly contains .wav files.
-
- Looks shallowly (the root itself and its immediate subdirectories) in
- both the audio.cpp checkout and the tts-audiobook-generator root (where
- audiobook.py lives), since clone reference .wavs commonly live in
- either. The tts-audiobook-generator ``output/`` directory is excluded.
- When exactly one candidate is found it is returned (as a starting
- directory for the .wav browser); when none or several are found None is
- returned so the caller falls back to its default start location.
- """
- candidates: List[Path] = []
- seen: Set[Path] = set()
-
- def consider(directory: Path) -> None:
- try:
- resolved = directory.resolve()
- except OSError:
- return
- if resolved in seen:
- return
- seen.add(resolved)
- if count_wavs(directory) > 0:
- candidates.append(directory)
-
- for root in (audiocpp_dir, tts_root):
- if not root.is_dir():
- continue
- consider(root)
- try:
- children = sorted(root.iterdir(), key=lambda p: p.name.lower())
- except OSError:
- continue
- for child in children:
- if not child.is_dir() or child.name.startswith("."):
- continue
- if root == tts_root and child.name == TTS_OUTPUT_DIR:
- continue
- consider(child)
-
- if len(candidates) == 1:
- return candidates[0]
- return None
-
-
-def wav_dir_info(directory: Path) -> Tuple[str, str]:
- """TUI status describing the directory listed in the wav browser."""
- count = count_wavs(directory)
- if count:
- wavs = ".wav" if count == 1 else ".wavs"
- return (f"{count} {wavs} found in this directory. Press Enter.",
- "ok")
- return ("No .wav files found in this directory", "warn")
-
-
-def wav_dir_preview(directory: Path) -> Tuple[str, str]:
- """TUI status describing a highlighted subdirectory in the wav browser."""
- count = count_wavs(directory)
- if count:
- wavs = ".wav" if count == 1 else ".wavs"
- return (f"{count} {wavs}", "ok")
- return ("no .wav files", "info")
-
-
-def url_with_port(url: str, port: int) -> str:
- """Return URL with its port replaced/inserted as PORT."""
- parts = urllib.parse.urlsplit(url)
- host = parts.hostname or "127.0.0.1"
- return urllib.parse.urlunsplit(
- (parts.scheme or "http", f"{host}:{port}", parts.path, "", ""))
-
-
-def server_running(url: str, timeout: float = 0.3) -> bool:
- """True when something accepts TCP connections at URL's host:port.
-
- A protocol-agnostic socket connect: an HTTP TTS server that is up will
- accept the connection (we do not need to speak HTTP to know it is
- listening). Returns False on any parse or connection error, so a
- misconfigured URL never blocks the hub — it just reports the backend
- as not running. Used by each backend's ``detect()`` to set
- ``BackendStatus.running``.
- """
- import socket
- try:
- parts = urllib.parse.urlsplit(url)
- host = parts.hostname or "127.0.0.1"
- port = parts.port or (443 if (parts.scheme or "http") == "https"
- else 80)
- except ValueError:
- return False
- try:
- with socket.create_connection((host, port), timeout=timeout):
- return True
- except OSError:
- return False
-
-
-def update_config_value(key: str, value: str,
- config_path: Optional[Path] = None) -> bool:
- """Rewrite a ``KEY = "value"`` line in converter/config.py.
-
- Only the quoted literal is replaced; surrounding lines and the trailing
- comment are preserved. Returns True when the file was changed. Used by
- the qwen and faster wizards to keep their API URL / voice / speaker
- settings in sync with the converter.
- """
- path = Path(config_path) if config_path is not None else CONFIG_PATH
- try:
- text = path.read_text(encoding="utf-8")
- except OSError:
- return False
- match = re.search(r'(?m)^(\s*' + re.escape(key) + r'\s*=\s*")([^"]*)(")',
- text)
- if not match or match.group(2) == value:
- return False
- text = text[:match.start(2)] + value + text[match.end(2):]
- try:
- path.write_text(text, encoding="utf-8")
- except OSError:
- return False
- return True
-
-
-def read_prompt_text(prompt_path: Path) -> Dict[str, str]:
- """Parse a prompt_text file into a stem -> transcript mapping.
-
- Lines are ``<name>|<transcript>``; blank lines are skipped and a line
- without a ``|`` separator is treated as a name with an empty transcript.
- Returns an empty mapping when the file does not exist.
- """
- if not prompt_path.exists():
- return {}
- mapping: Dict[str, str] = {}
- for line in prompt_path.read_text(encoding="utf-8").splitlines():
- if not line.strip():
- continue
- if "|" in line:
- name, _, text = line.partition("|")
- else:
- name, text = line, ""
- mapping[name.strip()] = text
- return mapping
-
-
-def write_prompt_text(wav_dir: Path,
- transcripts: Dict[str, str]) -> Path:
- """Write the voice_dir prompt_text mapping into WAV_DIR.
-
- One ``<basename-without-extension>|<transcript>`` line per voice.
- Returns the path of the written file.
- """
- prompt_path = wav_dir / PROMPT_TEXT_FILENAME
- lines = [f"{name}|{text}" for name, text in transcripts.items()]
- prompt_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
- return prompt_path
-
-
-def run_console_subprocess(argv: List[str], cwd: Optional[Path] = None) -> int:
- """Run a subprocess whose output streams to the plain console.
-
- Used inside ``tui.suspend`` for clone/build/pip steps: the caller has
- already left curses mode, so the child inherits the real terminal and
- its output appears normally. Returns the process exit code.
- """
- import subprocess
- try:
- result = subprocess.run(argv, cwd=str(cwd) if cwd is not None else None)
- except OSError as exc:
- print(f"[ERROR] Could not run {' '.join(argv)}: {exc}")
- return 1
- return result.returncode
-
-
-def git_clone(url: str, target: Path) -> int:
- """Clone URL into TARGET, streaming to the console. Returns exit code."""
- print(f"[INFO] Cloning {url} into {target}...")
- return run_console_subprocess(["git", "clone", url, str(target)])
-
-
-def pip_install(packages: List[str]) -> int:
- """pip install PACKAGES into the managed venv (``envs/tts``). Returns exit code.
-
- Delegates to ``backends.envs.pip_install`` so backend TTS packages are
- installed alongside the app requirements in the tool-managed environment
- rather than into whatever interpreter happens to be running the wizard.
- The import is local to avoid a circular import (envs imports this module).
- """
- from backends import envs
- return envs.pip_install(packages)
diff --git a/backends/envs.py b/backends/envs.py
deleted file mode 100644
index 7596db6..0000000
--- a/backends/envs.py
+++ /dev/null
@@ -1,185 +0,0 @@
-"""The managed Python environment for the audiobook generator and its backends.
-
-audiobook.py is meant to be launched from any Python (a bare system interpreter
-is fine): on startup it bootstraps a single tool-managed venv at
-``envs/tts`` and re-execs itself inside it. That venv holds both the
-audiobook app's own ``requirements.txt`` dependencies and the backend TTS
-packages (``qwen-tts``, ``faster-qwen3-tts[demo]``) the setup wizards pip
-install, so nothing is ever installed into the launching interpreter's
-environment.
-
-A parent process never needs to "activate" an environment — activation is
-just a shell convenience that puts an env's ``bin`` on PATH. Instead every
-helper here resolves the env's binaries by absolute path
-(``envs/tts/bin/python``, ``envs/tts/bin/qwen-tts-demo``), so the hub can
-spawn servers in this env from any parent environment.
-
-This module is imported before audiobook.py's third-party dependencies, so
-it must stay stdlib-only (it may import ``backends.common``, which is also
-stdlib-only, but never ``converter`` or the backend modules).
-"""
-
-import hashlib
-import os
-import sys
-from pathlib import Path
-from typing import List
-
-from backends import common
-
-# The tts-audiobook-generator checkout root (where audiobook.py lives).
-TTS_ROOT = Path(__file__).resolve().parent.parent
-
-# One shared venv for the app requirements and every pip-installed backend.
-ENV_DIR = TTS_ROOT / "envs" / "tts"
-REQUIREMENTS_PATH = TTS_ROOT / "requirements.txt"
-
-# Marker file recording the requirements.txt hash last installed into the env,
-# so ensure_app_env() re-installs when requirements.txt changes.
-MARKER_PATH = ENV_DIR / ".audiobook_env_ready"
-
-
-def _is_windows() -> bool:
- return sys.platform == "win32"
-
-
-def env_python() -> Path:
- """Absolute path to the venv's python interpreter."""
- return ENV_DIR / ("Scripts/python.exe" if _is_windows() else "bin/python")
-
-
-def env_script(name: str) -> Path:
- """Absolute path to a console script installed in the venv (e.g. qwen-tts-demo)."""
- subdir = "Scripts" if _is_windows() else "bin"
- suffix = ".exe" if _is_windows() else ""
- return ENV_DIR / subdir / f"{name}{suffix}"
-
-
-def env_exists() -> bool:
- """True when the venv's python interpreter is present on disk."""
- return env_python().is_file()
-
-
-def is_managed_env() -> bool:
- """True when the current process is already running inside the managed venv."""
- try:
- return Path(sys.executable).resolve() == env_python().resolve()
- except OSError:
- return False
-
-
-def create_env() -> int:
- """Create the venv with the launching interpreter (inherits its version).
-
- pip is bootstrapped inside the venv by ensurepip. Returns the ``python -m
- venv`` exit code; a non-zero result is reported with platform remediation.
- """
- print(f"[INFO] creating managed environment at {ENV_DIR}...")
- rc = common.run_console_subprocess(
- [sys.executable, "-m", "venv", str(ENV_DIR)])
- if rc != 0:
- print(f"[ERROR] python -m venv failed (exit {rc}).")
- if _is_windows():
- print(" On Windows make sure the launcher has the venv module.")
- else:
- print(" On Debian/Ubuntu install the venv package, e.g.:")
- print(" sudo apt install python3-venv")
- return rc
-
-
-def install_requirements() -> int:
- """pip install -r requirements.txt into the venv. Returns pip's exit code."""
- print(f"[INFO] pip install -r {REQUIREMENTS_PATH} into {ENV_DIR}...")
- return common.run_console_subprocess(
- [str(env_python()), "-m", "pip", "install", "-r", str(REQUIREMENTS_PATH)])
-
-
-def pip_install(packages: List[str]) -> int:
- """pip install PACKAGES into the venv, creating it first if needed.
-
- Used by the qwen/faster setup wizards to install backend TTS packages
- alongside the app requirements. Returns pip's exit code.
- """
- if not env_exists() and create_env() != 0:
- return 1
- print(f"[INFO] pip install {' '.join(packages)} into {ENV_DIR}...")
- return common.run_console_subprocess(
- [str(env_python()), "-m", "pip", "install", *packages])
-
-
-def module_available(module: str) -> bool:
- """True when MODULE imports inside the venv (e.g. qwen_tts, faster_qwen3_tts).
-
- A short subprocess probe against the venv's interpreter — the equivalent of
- importlib.util.find_spec, but for the managed env rather than the current
- one. Used by each backend's ``_is_installed``.
- """
- if not env_exists():
- return False
- import subprocess
- try:
- result = subprocess.run(
- [str(env_python()), "-c", f"import {module}"],
- stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
- timeout=30, check=False)
- except (OSError, subprocess.TimeoutExpired):
- return False
- return result.returncode == 0
-
-
-def _requirements_sha() -> str:
- try:
- data = REQUIREMENTS_PATH.read_bytes()
- except OSError:
- return ""
- return hashlib.sha256(data).hexdigest()
-
-
-def _marker_valid() -> bool:
- try:
- return MARKER_PATH.read_text(encoding="utf-8").strip() == _requirements_sha()
- except OSError:
- return False
-
-
-def _write_marker() -> None:
- try:
- MARKER_PATH.write_text(_requirements_sha() + "\n", encoding="utf-8")
- except OSError:
- pass
-
-
-def ensure_app_env() -> None:
- """Make sure the venv exists and has the current requirements.txt installed.
-
- Creates the venv when missing, and (re)installs requirements.txt when it is
- missing or has changed since the last install (tracked by a hash marker).
- Raises RuntimeError on any failure so the caller can abort before re-exec.
- """
- if not env_exists() and create_env() != 0:
- raise RuntimeError("could not create the managed environment")
- if not _marker_valid():
- if install_requirements() != 0:
- raise RuntimeError("pip install -r requirements.txt failed")
- _write_marker()
-
-
-def bootstrap(script_path: str) -> None:
- """Run audiobook.py inside the managed venv, creating it first if needed.
-
- A no-op when the current process is already the venv's interpreter. Otherwise
- ensures the env (and requirements) are ready, then replaces the process with
- the venv's python running the same script and CLI args. Called at the top of
- audiobook.py before any third-party import.
- """
- if is_managed_env():
- return
- try:
- ensure_app_env()
- except RuntimeError as exc:
- print(f"[FATAL] {exc}", file=sys.stderr)
- sys.exit(1)
- py = str(env_python())
- target = str(Path(script_path).resolve())
- print(f"[INFO] re-launching inside managed environment: {py}")
- os.execv(py, [py, target, *sys.argv[1:]])
diff --git a/backends/faster.py b/backends/faster.py
deleted file mode 100755
index 50c6102..0000000
--- a/backends/faster.py
+++ /dev/null
@@ -1,416 +0,0 @@
-#!/usr/bin/env python3
-"""Set up the faster-qwen3-tts backend for the audiobook generator.
-
-faster-qwen3-tts is an OpenAI-compatible Qwen3-TTS server with CUDA-graph
-inference (NVIDIA GPU required). It always uses voice cloning, with the
-reference voice configured on the server through a ``voices.json``. This
-module sets the whole backend up end-to-end as a TUI: pip-install the
-package, clone the repo (for ``examples/openai_server.py``), build a
-``voices.json`` from a directory of .wav references (transcribed with
-Whisper), sync ``converter/config.py``, and print the launch command. It is
-driven by ``audiobook.py``'s hub but can also be run directly with flags.
-
-Usage:
- python -m backends.faster [--wavs WAV_DIR] [--output PATH]
- [--language LANG] [--whisper-model NAME] [--force]
- [--port PORT] [--voice NAME] [--skip-install] [--skip-clone]
-"""
-
-import argparse
-import json
-import sys
-from pathlib import Path
-from typing import List, Optional
-
-sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
-
-from backends import (
- BackendStatus,
- ConfigureAction,
- ServerSpec,
- common,
- envs,
- format_launch_hint,
-)
-from backends.common import (
- TTS_ROOT,
- VOICES_DIR,
- find_wav_files,
- normalize_dir_arg,
-)
-from converter import config
-from converter.tts import (
- normalize_language,
- transcribe_reference_audio,
- whisper_backend_available,
-)
-from ui import tui
-
-FASTER_DIR_NAME = "faster-qwen3-tts"
-FASTER_GIT_URL = "https://github.com/andimarafioti/faster-qwen3-tts"
-FASTER_PIP_PKG = "faster-qwen3-tts[demo]"
-WHISPER_MODELS = ("tiny", "base", "small", "medium", "large-v3")
-
-
-def _checkout() -> Path:
- return TTS_ROOT / FASTER_DIR_NAME
-
-
-def _is_installed() -> bool:
- return envs.module_available("faster_qwen3_tts")
-
-
-def _is_cloned() -> bool:
- return (_checkout() / "examples" / "openai_server.py").is_file()
-
-
-def _config_port() -> int:
- import urllib.parse
- try:
- return urllib.parse.urlsplit(config.FASTER_API_URL).port or 8000
- except ValueError:
- return 8000
-
-
-def build_voices(wav_files: list, language: str, whisper_model: str) -> dict:
- """Transcribe each wav file and build the voices mapping."""
- voices = {}
- for wav_file in wav_files:
- name = wav_file.stem
- print(f"[INFO] Transcribing {wav_file.name} (voice '{name}')...")
- text = transcribe_reference_audio(str(wav_file), model_name=whisper_model)
- if text:
- print(f"[OK] {name}: {text}")
- else:
- print(f"[WARNING] No transcript for '{name}'; the faster backend "
- "strongly recommends an accurate transcript — consider "
- "editing voices.json by hand before starting the server")
- voices[name] = {
- "ref_audio": str(wav_file.resolve()),
- "ref_text": text or "",
- "language": language,
- }
- return voices
-
-
-def _write_voices_json(output_path: Path, wav_dir: Path, language: str,
- whisper_model: str, force: bool) -> Optional[dict]:
- """Transcribe the wav dir and write voices.json; return the voices dict."""
- wav_files = find_wav_files(wav_dir)
- if not wav_files:
- print(f"[ERROR] No .wav files found in {wav_dir}")
- return None
- if whisper_backend_available() is None:
- print("[WARNING] Neither faster_whisper nor whisper was found, so "
- "transcripts will be empty — install one or edit voices.json "
- "by hand.")
- voices = build_voices(wav_files, language, whisper_model)
- with output_path.open("w", encoding="utf-8") as handle:
- json.dump(voices, handle, indent=4, ensure_ascii=False)
- handle.write("\n")
- print(f"[OK] Wrote {output_path} with {len(voices)} voice(s): "
- f"{', '.join(voices)}")
- return voices
-
-
-def _wizard(stdscr, args: argparse.Namespace) -> Optional[dict]:
- """Linear TUI wizard collecting every faster-setup decision."""
- _GO_BACK = object()
-
- def confirm(question: str, default: bool = True) -> Optional[bool]:
- res = tui.confirm(stdscr, question, default=default,
- cancel_value=_GO_BACK)
- return None if res is _GO_BACK else res
-
- # Step 0: pip install (if not installed and not skipped).
- do_install = False
- if not _is_installed() and not args.skip_install:
- choice = confirm("faster-qwen3-tts is not installed. "
- "pip install it now?", default=True)
- if choice is None:
- return None
- do_install = choice
-
- # Step 1: clone (if not cloned and not skipped).
- do_clone = False
- if not _is_cloned() and not args.skip_clone:
- choice = confirm(f"faster-qwen3-tts repo not cloned. Clone it into "
- f"./{FASTER_DIR_NAME}?", default=True)
- if choice is None:
- return None
- do_clone = choice
-
- # Step 2: voices.json — wav dir, language, whisper model, output path.
- wav_dir = args.input_dir
- if wav_dir is None:
- wav_dir = tui.browse_directory(
- stdscr, "Select the directory with your .wav voices",
- info=common.wav_dir_info, preview=common.wav_dir_preview,
- start=VOICES_DIR)
- language = args.language
- if language is None:
- lang_text = tui.line_edit(
- stdscr, "Language", config.LANGUAGE,
- validate=lambda s: None if _try_language(s)
- else "Unknown language (e.g. English, en)",
- help_lines=["Language for every voice, as passed to the TTS "
- "model (names or short codes accepted)"])
- language = lang_text
- whisper_model = args.whisper_model
- if whisper_model is None:
- whisper_model = tui.menu(
- stdscr, "Whisper model for transcription",
- [(m, m) for m in WHISPER_MODELS],
- default_index=WHISPER_MODELS.index("base"))
- output_path = args.output
- if output_path is None:
- # Default into the cloned checkout; fall back to the wav directory
- # when the checkout is not present (so a flag-only run still works).
- output_path = (_checkout() / "voices.json") if _is_cloned() \
- else (wav_dir / "voices.json")
- if output_path.exists() and not args.force:
- choice = confirm(f"{output_path} already exists. Overwrite?",
- default=True)
- if choice is None or choice is False:
- # Fall back to a path in the current directory.
- output_path = Path.cwd() / "voices.json"
-
- # Step 3: port + default voice.
- port = args.port
- if port is None:
- port_text = tui.line_edit(
- stdscr, "Server port", str(_config_port()),
- validate=lambda s: None if (s.isdigit() and 1 <= int(s) <= 65535)
- else "Enter a port number between 1 and 65535")
- port = int(port_text)
-
- return {
- "do_install": do_install,
- "do_clone": do_clone,
- "wav_dir": wav_dir,
- "language": language,
- "whisper_model": whisper_model,
- "output_path": output_path,
- "port": port,
- "force": args.force,
- }
-
-
-def _try_language(value: str) -> bool:
- try:
- normalize_language(value)
- return True
- except ValueError:
- return False
-
-
-def _execute(settings: dict) -> int:
- """Console tail: install, clone, write voices.json, sync, advise."""
- if settings["do_install"]:
- rc = common.pip_install([FASTER_PIP_PKG])
- if rc != 0:
- print(f"[WARNING] pip install failed (exit {rc}); install "
- f"{FASTER_PIP_PKG} manually")
- else:
- print("[OK] faster-qwen3-tts installed")
-
- if settings["do_clone"]:
- rc = common.git_clone(FASTER_GIT_URL, _checkout())
- if rc != 0:
- print(f"[WARNING] git clone failed (exit {rc}); clone manually: "
- f"git clone {FASTER_GIT_URL} {_checkout()}")
- else:
- print(f"[OK] cloned into {_checkout()}")
-
- voices = _write_voices_json(settings["output_path"], settings["wav_dir"],
- settings["language"], settings["whisper_model"],
- settings["force"])
- if voices is None:
- return 1
-
- # Sync converter/config.py port + default voice.
- port = settings["port"]
- new_url = common.url_with_port(config.FASTER_API_URL, port)
- if new_url != config.FASTER_API_URL:
- if common.update_config_value("FASTER_API_URL", new_url):
- print(f"[OK] Updated FASTER_API_URL to {new_url}")
- else:
- print("[WARNING] Could not update FASTER_API_URL; edit "
- "converter/config.py by hand")
- default_voice = next(iter(voices))
- if default_voice != config.FASTER_VOICE:
- if common.update_config_value("FASTER_VOICE", default_voice):
- print(f"[OK] Updated FASTER_VOICE to {default_voice}")
- else:
- print("[WARNING] Could not update FASTER_VOICE; edit "
- "converter/config.py by hand")
-
- _print_launch_hint(settings["output_path"], port)
- return 0
-
-
-def _print_launch_hint(voices_path: Path, port: int) -> None:
- print()
- if _is_cloned():
- py = envs.env_python()
- print("Start the server with (or use the hub's 'Server' menu):")
- print(f" {py} {_checkout()}/examples/openai_server.py "
- f"--voices {voices_path} --port {port}")
- else:
- print("[INFO] Clone faster-qwen3-tts to get examples/openai_server.py,")
- print(f" then run it with --voices {voices_path} --port {port}")
-
-
-def run_tui(args: Optional[argparse.Namespace] = None) -> int:
- """Run the faster setup wizard end-to-end."""
- import curses
- if args is None:
- args = build_parser().parse_args([])
- try:
- settings = curses.wrapper(_wizard, args)
- except tui.WizardCancelled:
- print("\n[INFO] Cancelled; nothing was written")
- return 1
- try:
- curses.curs_set(1)
- except curses.error:
- pass
- if settings is None:
- print("[INFO] Aborted")
- return 1
- return _execute(settings)
-
-
-def _collect_from_flags(args: argparse.Namespace,
- parser: argparse.ArgumentParser) -> Optional[dict]:
- """Build the settings dict from flags for a non-interactive run."""
- wav_dir = args.input_dir if args.input_dir is not None else VOICES_DIR
- if not wav_dir.is_dir():
- parser.error(f"WAV directory not found: {wav_dir}")
- try:
- language = normalize_language(args.language or config.LANGUAGE)
- except ValueError as exc:
- parser.error(str(exc))
- output_path = args.output if args.output is not None \
- else ((_checkout() / "voices.json") if _is_cloned()
- else (wav_dir / "voices.json"))
- if output_path.exists() and not args.force:
- print("[INFO] Aborted; existing voices.json kept")
- return None
- return {
- "do_install": (not _is_installed()) and not args.skip_install,
- "do_clone": (not _is_cloned()) and not args.skip_clone,
- "wav_dir": wav_dir,
- "language": language,
- "whisper_model": args.whisper_model or "base",
- "output_path": output_path,
- "port": args.port if args.port is not None else _config_port(),
- "force": args.force,
- }
-
-
-def build_parser() -> argparse.ArgumentParser:
- parser = argparse.ArgumentParser(
- description="Set up the faster-qwen3-tts backend: pip install, clone, "
- "build voices.json, and sync converter/config.py.")
- parser.add_argument("input_dir", type=normalize_dir_arg, nargs="?",
- default=None, metavar="WAV_DIR",
- help="Directory with .wav reference files "
- f"(default: {VOICES_DIR}; browsed for in the TUI)")
- parser.add_argument("--output", type=Path, default=None,
- help="Output path for voices.json (default: "
- "./faster-qwen3-tts/voices.json, or "
- "WAV_DIR/voices.json when not cloned)")
- parser.add_argument("--language", type=str, default=None,
- help="Language for all voices (default: English; "
- "names and short codes accepted)")
- parser.add_argument("--whisper-model", type=str, default=None,
- choices=WHISPER_MODELS,
- help="Whisper model size for transcription "
- "(default: base)")
- parser.add_argument("--force", action="store_true",
- help="Overwrite an existing voices.json without "
- "prompting")
- parser.add_argument("--port", type=int, default=None,
- help="Server port to record in converter/config.py "
- "(default: the port in FASTER_API_URL)")
- parser.add_argument("--skip-install", action="store_true",
- help="Do not pip install faster-qwen3-tts[demo]")
- parser.add_argument("--skip-clone", action="store_true",
- help="Do not clone the faster-qwen3-tts repo")
- return parser
-
-
-def detect() -> BackendStatus:
- """Detect how far faster-qwen3-tts is set up, plus the launch command."""
- installed = _is_installed()
- cloned = _is_cloned()
- voices_json = _checkout() / "voices.json"
- configured = installed and cloned and voices_json.exists()
- running = common.server_running(config.FASTER_API_URL)
- details: List[str] = []
- details.append("pip: installed" if installed else
- "not installed — run setup to pip install")
- details.append(f"checkout: {_checkout()}" if cloned else
- f"not cloned — run setup to clone ./{FASTER_DIR_NAME}")
- details.append(f"voices: {voices_json}" if voices_json.exists() else
- "no voices.json — run setup to create one")
- launch = ""
- servers: List[ServerSpec] = []
- if cloned and voices_json.exists():
- argv = [str(envs.env_python()),
- str(_checkout() / "examples" / "openai_server.py"),
- "--voices", str(voices_json), "--port", str(_config_port())]
- servers = [ServerSpec("faster", config.FASTER_API_URL, argv)]
- launch = format_launch_hint(servers)
- return BackendStatus("faster", "faster-qwen3-tts",
- installed=installed and cloned,
- configured=configured, running=running,
- details=details, launch_hint=launch,
- servers=servers)
-
-
-def _run_voices_only_tui() -> int:
- """Rebuild voices.json via the TUI (the "configure" action).
-
- Runs the same wizard but skips the pip/clone prerequisites so it goes
- straight to picking the .wav directory and writing voices.json.
- """
- args = build_parser().parse_args([])
- args.skip_install = True
- args.skip_clone = True
- return run_tui(args)
-
-
-configure_actions: List[ConfigureAction] = [
- ConfigureAction("Rebuild voices.json", _run_voices_only_tui),
- ConfigureAction("Reconfigure faster-qwen3-tts", run_tui),
-]
-
-
-def main() -> int:
- parser = build_parser()
- args = parser.parse_args()
-
- if _interactive():
- return run_tui(args)
-
- settings = _collect_from_flags(args, parser)
- if settings is None:
- return 1
- return _execute(settings)
-
-
-def _interactive() -> bool:
- try:
- import curses # noqa: F401
- except ImportError:
- return False
- try:
- return sys.stdin.isatty() and sys.stdout.isatty()
- except (AttributeError, ValueError):
- return False
-
-
-if __name__ == "__main__":
- sys.exit(main())
diff --git a/backends/qwen.py b/backends/qwen.py
deleted file mode 100644
index 52f7a3f..0000000
--- a/backends/qwen.py
+++ /dev/null
@@ -1,272 +0,0 @@
-#!/usr/bin/env python3
-"""Set up the Qwen3-TTS demo backend for the audiobook generator.
-
-qwen-tts is a pip package providing the ``qwen-tts-demo`` server, which
-hosts the Qwen3-TTS CustomVoice (built-in speakers) and Base (voice
-cloning) models on separate ports. This module sets it up end-to-end as a
-TUI: pip-install the package, configure the two ports and the built-in
-speaker in ``converter/config.py``, and print the launch commands. It is
-driven by ``audiobook.py``'s hub but can also be run directly with flags.
-
-Usage:
- python -m backends.qwen [--port-custom PORT] [--port-clone PORT]
- [--speaker NAME] [--skip-install]
-"""
-
-import argparse
-import sys
-from pathlib import Path
-from typing import List, Optional
-
-sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
-
-from backends import (
- BackendStatus,
- ConfigureAction,
- ServerSpec,
- common,
- envs,
- format_launch_hint,
-)
-from converter import config
-from ui import tui
-
-QWEN_PIP_PKG = "qwen-tts"
-QWEN_CUSTOMVOICE_MODEL = "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice"
-QWEN_BASE_MODEL = "Qwen/Qwen3-TTS-12Hz-1.7B-Base"
-DEFAULT_CUSTOM_PORT = 7860
-DEFAULT_CLONE_PORT = 7861
-
-# Built-in CustomVoice speakers (see converter/config.py SPEAKER).
-QWEN_SPEAKERS = ("Vivian", "Serena", "Uncle_Fu", "Dylan", "Eric", "Ryan",
- "Aiden", "Ono_Anna", "Sohee")
-
-
-def _is_installed() -> bool:
- if envs.env_script("qwen-tts-demo").is_file():
- return True
- return envs.module_available("qwen_tts")
-
-
-def _config_port(url: str, fallback: int) -> int:
- import urllib.parse
- try:
- return urllib.parse.urlsplit(url).port or fallback
- except ValueError:
- return fallback
-
-
-def _wizard(stdscr, args: argparse.Namespace) -> Optional[dict]:
- """Linear TUI wizard collecting every qwen-setup decision."""
- _GO_BACK = object()
-
- def confirm(question: str, default: bool = True) -> Optional[bool]:
- res = tui.confirm(stdscr, question, default=default,
- cancel_value=_GO_BACK)
- return None if res is _GO_BACK else res
-
- # Step 0: pip install (if not installed and not skipped).
- do_install = False
- if not _is_installed() and not args.skip_install:
- choice = confirm("qwen-tts is not installed. pip install it now?",
- default=True)
- if choice is None:
- return None
- do_install = choice
-
- # Step 1: ports.
- custom_port = args.port_custom
- if custom_port is None:
- port_text = tui.line_edit(
- stdscr, "CustomVoice (built-in speaker) port",
- str(_config_port(config.QWEN_API_URL, DEFAULT_CUSTOM_PORT)),
- validate=lambda s: None if (s.isdigit() and 1 <= int(s) <= 65535)
- else "Enter a port number between 1 and 65535",
- help_lines=["The port for qwen-tts-demo CustomVoice (speaker mode)"])
- custom_port = int(port_text)
- clone_port = args.port_clone
- if clone_port is None:
- port_text = tui.line_edit(
- stdscr, "Base (voice clone) port",
- str(_config_port(config.CLONE_API_URL, DEFAULT_CLONE_PORT)),
- validate=lambda s: None if (s.isdigit() and 1 <= int(s) <= 65535)
- else "Enter a port number between 1 and 65535",
- help_lines=["The port for qwen-tts-demo Base (voice cloning)"])
- clone_port = int(port_text)
-
- # Step 2: built-in speaker.
- speaker = args.speaker
- if speaker is None:
- speaker = tui.menu(
- stdscr, "Built-in CustomVoice speaker",
- [(s, s) for s in QWEN_SPEAKERS],
- default_index=max(0, QWEN_SPEAKERS.index(config.SPEAKER)
- if config.SPEAKER in QWEN_SPEAKERS else 0),
- help_lines=["Used by audiobook.py --backend qwen without --clone"])
-
- return {
- "do_install": do_install,
- "custom_port": custom_port,
- "clone_port": clone_port,
- "speaker": speaker,
- }
-
-
-def _execute(settings: dict) -> int:
- """Console tail: install, sync config, advise."""
- if settings["do_install"]:
- rc = common.pip_install([QWEN_PIP_PKG])
- if rc != 0:
- print(f"[WARNING] pip install failed (exit {rc}); install "
- f"{QWEN_PIP_PKG} manually")
- else:
- print(f"[OK] {QWEN_PIP_PKG} installed")
-
- custom_url = common.url_with_port(config.QWEN_API_URL, settings["custom_port"])
- if custom_url != config.QWEN_API_URL:
- if common.update_config_value("QWEN_API_URL", custom_url):
- print(f"[OK] Updated QWEN_API_URL to {custom_url}")
- else:
- print("[WARNING] Could not update QWEN_API_URL; edit "
- "converter/config.py by hand")
- clone_url = common.url_with_port(config.CLONE_API_URL, settings["clone_port"])
- if clone_url != config.CLONE_API_URL:
- if common.update_config_value("CLONE_API_URL", clone_url):
- print(f"[OK] Updated CLONE_API_URL to {clone_url}")
- else:
- print("[WARNING] Could not update CLONE_API_URL; edit "
- "converter/config.py by hand")
- if settings["speaker"] != config.SPEAKER:
- if common.update_config_value("SPEAKER", settings["speaker"]):
- print(f"[OK] Updated SPEAKER to {settings['speaker']}")
- else:
- print("[WARNING] Could not update SPEAKER; edit "
- "converter/config.py by hand")
-
- _print_launch_hint(settings["custom_port"], settings["clone_port"])
- return 0
-
-
-def _print_launch_hint(custom_port: int, clone_port: int) -> None:
- demo = envs.env_script("qwen-tts-demo")
- print()
- print("Start the servers (in separate terminals), or use the hub's")
- print("'Server' menu / let a conversion start one automatically:")
- print(f" {demo} {QWEN_CUSTOMVOICE_MODEL} --ip 127.0.0.1 "
- f"--port {custom_port}")
- print(f" {demo} {QWEN_BASE_MODEL} --ip 127.0.0.1 "
- f"--port {clone_port}")
- print("Then run: python audiobook.py --backend qwen")
-
-
-def run_tui(args: Optional[argparse.Namespace] = None) -> int:
- """Run the qwen setup wizard end-to-end."""
- import curses
- if args is None:
- args = build_parser().parse_args([])
- try:
- settings = curses.wrapper(_wizard, args)
- except tui.WizardCancelled:
- print("\n[INFO] Cancelled; nothing was written")
- return 1
- try:
- curses.curs_set(1)
- except curses.error:
- pass
- if settings is None:
- print("[INFO] Aborted")
- return 1
- return _execute(settings)
-
-
-def _collect_from_flags(args: argparse.Namespace,
- parser: argparse.ArgumentParser) -> dict:
- return {
- "do_install": (not _is_installed()) and not args.skip_install,
- "custom_port": args.port_custom if args.port_custom is not None
- else _config_port(config.QWEN_API_URL, DEFAULT_CUSTOM_PORT),
- "clone_port": args.port_clone if args.port_clone is not None
- else _config_port(config.CLONE_API_URL, DEFAULT_CLONE_PORT),
- "speaker": args.speaker or config.SPEAKER,
- }
-
-
-def build_parser() -> argparse.ArgumentParser:
- parser = argparse.ArgumentParser(
- description="Set up the Qwen3-TTS demo backend: pip install, "
- "configure ports/speaker, and print launch commands.")
- parser.add_argument("--port-custom", type=int, default=None,
- help="CustomVoice (speaker) port (default: "
- f"{DEFAULT_CUSTOM_PORT})")
- parser.add_argument("--port-clone", type=int, default=None,
- help="Base (voice clone) port (default: "
- f"{DEFAULT_CLONE_PORT})")
- parser.add_argument("--speaker", type=str, default=None,
- choices=QWEN_SPEAKERS,
- help="Built-in CustomVoice speaker (default: "
- f"{config.SPEAKER})")
- parser.add_argument("--skip-install", action="store_true",
- help="Do not pip install qwen-tts")
- return parser
-
-
-def detect() -> BackendStatus:
- """Detect whether qwen-tts is installed, plus the launch commands."""
- installed = _is_installed()
- custom_port = _config_port(config.QWEN_API_URL, DEFAULT_CUSTOM_PORT)
- clone_port = _config_port(config.CLONE_API_URL, DEFAULT_CLONE_PORT)
- # Running when either server is up — CustomVoice (speaker mode) or Base
- # (voice clone) each suffice for a conversion on their own.
- running = (common.server_running(config.QWEN_API_URL)
- or common.server_running(config.CLONE_API_URL))
- details: List[str] = []
- details.append("pip: installed" if installed else
- "not installed — run setup to pip install qwen-tts")
- details.append(f"CustomVoice port: {custom_port}")
- details.append(f"Base (clone) port: {clone_port}")
- details.append(f"speaker: {config.SPEAKER}")
- demo = str(envs.env_script("qwen-tts-demo"))
- servers = [
- ServerSpec("qwen-custom", config.QWEN_API_URL,
- [demo, QWEN_CUSTOMVOICE_MODEL, "--ip", "127.0.0.1",
- "--port", str(custom_port)]),
- ServerSpec("qwen-clone", config.CLONE_API_URL,
- [demo, QWEN_BASE_MODEL, "--ip", "127.0.0.1",
- "--port", str(clone_port)]),
- ]
- return BackendStatus("qwen", "qwen-tts",
- installed=installed, configured=installed,
- running=running, details=details,
- launch_hint=format_launch_hint(servers),
- servers=servers)
-
-
-configure_actions: List[ConfigureAction] = [
- ConfigureAction("Reconfigure qwen-tts (ports/speaker)", run_tui),
-]
-
-
-def main() -> int:
- parser = build_parser()
- args = parser.parse_args()
-
- if _interactive():
- return run_tui(args)
-
- settings = _collect_from_flags(args, parser)
- return _execute(settings)
-
-
-def _interactive() -> bool:
- try:
- import curses # noqa: F401
- except ImportError:
- return False
- try:
- return sys.stdin.isatty() and sys.stdout.isatty()
- except (AttributeError, ValueError):
- return False
-
-
-if __name__ == "__main__":
- sys.exit(main())
diff --git a/backends/servers.py b/backends/servers.py
deleted file mode 100644
index 12846f0..0000000
--- a/backends/servers.py
+++ /dev/null
@@ -1,244 +0,0 @@
-"""Start and stop TTS backend servers from the TUI hub.
-
-Each backend's ``detect()`` returns a list of ``ServerSpec`` — the exact argv
-(absolute binaries in the managed venv, no shell activation needed) and the
-URL to probe for readiness. This module turns those specs into running
-processes: ``start`` spawns the server, streams its output to
-``logs/<name>-server.log``, records its pid, and polls the URL until it
-accepts connections (model loads are slow, so the timeout is generous);
-``stop`` terminates the process group the hub started.
-
-Everything here runs in the plain console tail after the curses TUI returns
-(matching the wizards' build/pip streaming), so progress and log tails appear
-normally. Pid/log files live under ``logs/`` which is already gitignored.
-"""
-
-import os
-import signal
-import subprocess
-import sys
-import time
-from pathlib import Path
-from typing import List
-
-from backends import common
-from backends.common import TTS_ROOT
-
-LOG_DIR = TTS_ROOT / "logs"
-
-# How long to wait for a server to accept connections on its URL. First-time
-# model loads (especially qwen-tts / faster-qwen3-tts pulling weights into
-# VRAM) can take minutes, so this is deliberately generous.
-SERVER_START_TIMEOUT = 600
-
-# Grace period after SIGTERM before escalating to SIGKILL (POSIX).
-STOP_GRACE_SECONDS = 10
-
-
-def _log_path(name: str) -> Path:
- return LOG_DIR / f"{name}-server.log"
-
-
-def _pid_path(name: str) -> Path:
- return LOG_DIR / f"{name}-server.pid"
-
-
-def _tail_log(name: str, lines: int = 20) -> None:
- """Print the last LINES of the server's log (best-effort)."""
- path = _log_path(name)
- try:
- text = path.read_text(encoding="utf-8", errors="replace")
- except OSError:
- return
- tail = "\n".join(text.splitlines()[-lines:])
- if tail:
- print(f"--- last {lines} lines of {path} ---")
- print(tail)
- print("---")
-
-
-def _pid_alive(pid: int) -> bool:
- """True when a process with PID is still running (POSIX signal-0 probe)."""
- if sys.platform == "win32":
- try:
- import ctypes
- kernel32 = ctypes.windll.kernel32 # type: ignore[attr-defined]
- PROCESS_QUERY_LIMITED_INFORMATION = 0x1000
- handle = kernel32.OpenProcess(
- PROCESS_QUERY_LIMITED_INFORMATION, False, pid)
- if not handle:
- return False
- kernel32.CloseHandle(handle)
- return True
- except OSError:
- return False
- try:
- os.kill(pid, 0)
- except ProcessLookupError:
- return False
- except PermissionError:
- return True
- return True
-
-
-def _kill_pid(pid: int) -> bool:
- """Terminate PID (and its process group on POSIX). Returns True when dead."""
- if sys.platform == "win32":
- try:
- os.kill(pid, signal.SIGTERM)
- except (ProcessLookupError, PermissionError, OSError):
- return not _pid_alive(pid)
- for _ in range(int(STOP_GRACE_SECONDS * 10)):
- if not _pid_alive(pid):
- return True
- time.sleep(0.1)
- try:
- os.kill(pid, signal.SIGTERM)
- except OSError:
- pass
- return not _pid_alive(pid)
- # POSIX: kill the whole process group (started with start_new_session=True).
- try:
- pgid = os.getpgid(pid)
- except ProcessLookupError:
- return True
- try:
- os.killpg(pgid, signal.SIGTERM)
- except ProcessLookupError:
- return True
- except PermissionError:
- return False
- for _ in range(int(STOP_GRACE_SECONDS * 10)):
- try:
- os.killpg(pgid, 0)
- except ProcessLookupError:
- return True
- except PermissionError:
- return False
- time.sleep(0.1)
- try:
- os.killpg(pgid, signal.SIGKILL)
- except (ProcessLookupError, PermissionError):
- pass
- return True
-
-
-def start(spec) -> bool:
- """Start the server described by SPEC (a ``backends.ServerSpec``).
-
- Spawns its argv with stdout/stderr to ``logs/<name>-server.log``, records
- the pid, and polls ``common.server_running(spec.url)`` until it accepts
- connections or ``SERVER_START_TIMEOUT`` elapses. Returns True when the
- server is up; on timeout or early exit, prints the log tail and returns
- False. A no-op (True) when the server is already running.
- """
- argv: List[str] = list(spec.argv)
- exe = Path(argv[0])
- if not exe.exists():
- print(f"[ERROR] server executable not found: {exe}")
- print(" run 'Set up a backend' for "
- f"{spec.name!r} first.")
- return False
- if common.server_running(spec.url):
- print(f"[INFO] {spec.name} server already running on {spec.url}")
- return True
-
- LOG_DIR.mkdir(parents=True, exist_ok=True)
- pid_file = _pid_path(spec.name)
- if pid_file.exists():
- try:
- pid_file.unlink()
- except OSError:
- pass
-
- print(f"[INFO] starting {spec.name} server: "
- + " ".join(str(a) for a in argv))
- log_handle = _log_path(spec.name).open("w", encoding="utf-8")
- popen_kwargs = {"stdout": log_handle, "stderr": subprocess.STDOUT}
- if sys.platform == "win32":
- popen_kwargs["creationflags"] = \
- subprocess.CREATE_NEW_PROCESS_GROUP # type: ignore[attr-defined]
- else:
- popen_kwargs["start_new_session"] = True
- try:
- proc = subprocess.Popen(argv, **popen_kwargs)
- except OSError as exc:
- print(f"[ERROR] could not start server: {exc}")
- log_handle.close()
- return False
-
- pid_file.write_text(str(proc.pid), encoding="utf-8")
- print(f"[INFO] pid {proc.pid}; logs: {_log_path(spec.name)}")
-
- deadline = time.time() + SERVER_START_TIMEOUT
- while time.time() < deadline:
- if proc.poll() is not None:
- print(f"[ERROR] {spec.name} server exited with code "
- f"{proc.returncode}")
- _tail_log(spec.name)
- try:
- pid_file.unlink()
- except OSError:
- pass
- return False
- if common.server_running(spec.url):
- print(f"[OK] {spec.name} server is up on {spec.url}")
- return True
- time.sleep(1)
- print(f"[ERROR] {spec.name} server did not start within "
- f"{SERVER_START_TIMEOUT}s")
- _tail_log(spec.name)
- # Leave the pid file in place so stop() can kill it (it may still load).
- return False
-
-
-def stop(name: str) -> bool:
- """Stop a server previously started by ``start`` (identified by pid file).
-
- Returns True when the process was terminated (or already gone). Returns
- False when there is no pid file — the server was not started by this tool,
- so the user must stop it manually (e.g. close its terminal).
- """
- pid_file = _pid_path(name)
- if not pid_file.exists():
- print(f"[INFO] no pid file for '{name}' "
- "(not started by this tool — stop it manually)")
- return False
- try:
- pid = int(pid_file.read_text(encoding="utf-8").strip())
- except (OSError, ValueError):
- print(f"[WARNING] could not read pid file {pid_file}; removing it")
- try:
- pid_file.unlink()
- except OSError:
- pass
- return False
- if not _pid_alive(pid):
- print(f"[INFO] {name} server (pid {pid}) already stopped")
- try:
- pid_file.unlink()
- except OSError:
- pass
- return True
- print(f"[INFO] stopping {name} server (pid {pid})...")
- killed = _kill_pid(pid)
- if killed:
- print(f"[OK] {name} server stopped")
- else:
- print(f"[WARNING] could not stop pid {pid}; stop it manually")
- try:
- pid_file.unlink()
- except OSError:
- pass
- return killed
-
-
-def pid_for(name: str):
- """Return the recorded pid for NAME, or None when no pid file exists."""
- pid_file = _pid_path(name)
- if not pid_file.exists():
- return None
- try:
- return int(pid_file.read_text(encoding="utf-8").strip())
- except (OSError, ValueError):
- return None