#!/usr/bin/env python3 """Set up the audio.cpp TTS backend for the audiobook generator. This does the whole audio.cpp setup end-to-end as a full-screen DOS-style TUI: locate or clone an audio.cpp checkout into ``app/audio.cpp``, optionally build ``audiocpp_server``, pick model families/packages from the checkout's ``model_specs`` catalog, transcribe reference .wav voices, write ``server.json``, sync ``app/converter/config.py``, download the models, and print the exact command to start the server. It is driven by ``audiobook.py``'s TUI hub (``backends.REGISTRY``) but can also be run directly for scripting — every value has a flag, and a non-interactive run with all flags supplied never opens the TUI. The converter is family-agnostic (it detects the family of the selected entry from ``GET /v1/models`` at startup), so any TTS family listed in the catalog works without further changes. Usage: python app/backends/audiocpp.py [--wavs WAV_DIR] [--output PATH] [--audiocpp-dir PATH] [--clone] [--families FAM1,FAM2] [--all-packages] [--host HOST] [--port PORT] [--build-backend {cuda,vulkan,hip,cpu}] [--backend {cuda,vulkan,hip,cpu}] [--lazy-load] [--whisper-model NAME] [--force] [--download] [--no-sync-port] [--no-sync-model-ids] With no flags and a terminal, the TUI wizard runs. Without a terminal (or with all flags supplied), it runs non-interactively from the flags; any missing required value is a hard error with a remediation hint. """ import argparse import json import os import re import subprocess import sys import urllib.parse import urllib.request from pathlib import Path from typing import Callable, Dict, List, Optional, Set, Tuple # Allow running directly (python app/backends/audiocpp.py) from any cwd. sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) from backends import ( BackendStatus, ConfigureAction, ServerSpec, common, format_launch_hint, servers, ) from backends.common import ( APP_DIR, CONFIG_PATH, PROMPT_TEXT_FILENAME, TTS_ROOT, VOICES_DIR, detect_wav_dir, find_wav_files, normalize_dir_arg, read_prompt_text, resolve_wav_dir_arg, write_prompt_text, ) from backends.common import ( wav_dir_info as _wav_dir_info, ) from backends.common import ( wav_dir_preview as _wav_dir_preview, ) from converter import config from converter.tts import transcribe_reference_audio, whisper_backend_available from ui import tui DEFAULT_HOST = "127.0.0.1" FALLBACK_PORT = 8080 BACKENDS = ("cuda", "vulkan", "hip", "cpu") TASK_TTS = "tts" TASK_VDES = "vdes" # audio.cpp is cloned into a sibling directory of the audiobook generator. AUDIOCPP_DIR_NAME = "audio.cpp" AUDIOCPP_GIT_URL = "https://github.com/0xShug0/audio.cpp" # Sentinel returned by tui.confirm (via its cancel_value) when the user # presses Esc on an overwrite prompt to go back to the checkout browser # instead of aborting the wizard. _GO_BACK = object() class _GoBack(Exception): """Raised inside the TUI wizard to fall back to the previous screen group. Every wizard widget is passed ``back_value=_GO_BACK`` so Esc returns the sentinel instead of aborting; pickers and confirmations that call into callbacks (task/id pickers, the transcription plan, the download prompt) convert that sentinel into this exception so the enclosing step can catch it and step back. Only the first screen (the checkout browser) lets Esc abort the whole wizard. """ # Package names that mark a voice-design model (hosted with task "vdes"). DESIGN_PACKAGE_RE = re.compile(r"voice[\s_\-]?design", re.IGNORECASE) # Short, friendly default entry ids for selected families. Other families # derive an id from their family name (see default_model_id). All families # are listed equally, in alphabetical order. PREFERRED_IDS = { "qwen3_tts": "qwen", "higgs_audio_tts": "higgs", "voxcpm2": "voxcpm2", "index_tts2": "indextts2", } class _TuiError(Exception): """A fatal error raised from inside the TUI wizard. The message is reported to stderr after the terminal is restored; the process exits with code 2 (matching a parser error). """ def _interactive() -> bool: """True when the TUI wizard can run (curses importable + tty).""" try: import curses # noqa: F401 except ImportError: return False try: return sys.stdin.isatty() and sys.stdout.isatty() except (AttributeError, ValueError): return False def _resolve_audiocpp_root(directory: Path) -> Optional[Path]: """Return the audio.cpp checkout root for DIRECTORY, or None. Accepts either the checkout root itself (it must contain a ``model_specs`` directory) or the ``model_specs`` directory inside it (the parent is used), so the file browser cannot pick the wrong one of the two. """ if (directory / "model_specs").is_dir(): return directory if directory.name == "model_specs" and directory.is_dir(): return directory.parent return None def _audiocpp_root_status(directory: Path) -> Tuple[str, str]: """TUI status describing the directory listed in the checkout browser.""" if _resolve_audiocpp_root(directory) is not None: return ("model_specs/ found here", "ok") return ("No model_specs/ directory here", "warn") def _audiocpp_root_preview(directory: Path) -> Optional[Tuple[str, str]]: """TUI status for a highlighted subdirectory in the checkout browser.""" if (directory / "model_specs").is_dir(): return ("contains model_specs/", "ok") return None def _checkout_auto_select(entry: Path) -> Optional[Path]: """Auto-accept a highlighted checkout in the TUI browser. A subdirectory named ``audio.cpp`` that already contains a ``model_specs`` directory is the audio.cpp checkout root, so it is accepted immediately on Enter/Right (as if ``[ Use this directory ]`` had been pressed) instead of being descended into. Anything else returns None so the user keeps browsing. This is only consulted while auto-accepting is still enabled; after the user presses Esc to go back, the browser is restarted inside the previously accepted checkout and this callback is no longer passed, so a wrong guess can be corrected. """ if entry.name == "audio.cpp" and (entry / "model_specs").is_dir(): return entry return None # Backend display order, with short descriptions. The backend name is padded # so the descriptions' dashes line up in the menu. _BACKEND_DESCRIPTIONS = ( ("cuda", "NVIDIA GPUs (fastest)"), ("vulkan", "cross-vendor GPU"), ("hip", "AMD GPUs"), ("cpu", "no GPU required"), ) def _backend_options(detected: Optional[str] = None ) -> Tuple[List[Tuple[str, str]], int]: """Build the aligned backend menu options and the default index. The backend names are padded to a common width so the ``-`` dashes before the descriptions line up. When DETECTED matches one of the options, that option gets ``[auto-detected]`` appended and is the default (cursor/start) selection; otherwise the first option is the default as before. Returns (options, default_index). """ width = max(len(name) for name, _ in _BACKEND_DESCRIPTIONS) options: List[Tuple[str, str]] = [] default_index = 0 for index, (name, desc) in enumerate(_BACKEND_DESCRIPTIONS): label = f"{name.ljust(width)} - {desc}" if detected == name: label += " [auto-detected]" default_index = index options.append((label, name)) return options, default_index def config_port() -> int: """Return the port of AUDIOCPP_API_URL in app/converter/config.py.""" try: return urllib.parse.urlsplit(config.AUDIOCPP_API_URL).port or FALLBACK_PORT except ValueError: return FALLBACK_PORT def _url_with_port(url: str, port: int) -> str: parts = urllib.parse.urlsplit(url) host = parts.hostname or "127.0.0.1" return urllib.parse.urlunsplit( (parts.scheme or "http", f"{host}:{port}", parts.path, "", "")) def update_config_api_url_port(port: int, config_path: Optional[Path] = None) -> bool: """Rewrite the port inside AUDIOCPP_API_URL in app/converter/config.py. Only the quoted URL literal is replaced; surrounding lines and the trailing comment are preserved. Returns True when the file was changed. """ path = Path(config_path) if config_path is not None else CONFIG_PATH try: text = path.read_text(encoding="utf-8") except OSError: return False match = re.search(r'(?m)^(\s*AUDIOCPP_API_URL\s*=\s*")([^"]*)(")', text) if not match: return False new_url = _url_with_port(match.group(2), port) if new_url == match.group(2): return False text = text[:match.start(2)] + new_url + text[match.end(2):] try: path.write_text(text, encoding="utf-8") except OSError: return False return True def update_server_config_port(port: int) -> bool: """Rewrite the 'port' in the audio.cpp checkout's server.json. Loads ``/server.json``, sets its ``port`` to PORT, and rewrites it with the same ``json.dump`` formatting the wizard uses. Returns True when the file now carries PORT (a no-op when it already does), and False when there is no checkout/server.json or the file cannot be read or written. """ checkout = find_local_checkout() if checkout is None: return False server_json = checkout / "server.json" if not server_json.exists(): return False try: data = json.loads(server_json.read_text(encoding="utf-8")) except (OSError, ValueError): return False if not isinstance(data, dict): return False if data.get("port") == port: return True data["port"] = port try: with server_json.open("w", encoding="utf-8") as handle: json.dump(data, handle, indent=2, ensure_ascii=False) handle.write("\n") except OSError: return False return True def update_config_model_ids(model_id: str, clone_model_id: Optional[str] = None, config_path: Optional[Path] = None) -> bool: """Rewrite AUDIOCPP_MODEL_ID (and AUDIOCPP_CLONE_MODEL_ID when given). Only the quoted id literals are replaced; surrounding lines and comments are preserved. Returns True when the file was changed. """ path = Path(config_path) if config_path is not None else CONFIG_PATH try: text = path.read_text(encoding="utf-8") except OSError: return False updates: List[Tuple[str, str]] = [("AUDIOCPP_MODEL_ID", model_id)] if clone_model_id is not None: updates.append(("AUDIOCPP_CLONE_MODEL_ID", clone_model_id)) changed = False for name, value in updates: match = re.search(r'(?m)^(\s*' + name + r'\s*=\s*")([^"]*)(")', text) if match and match.group(2) != value: text = text[:match.start(2)] + value + text[match.end(2):] changed = True if not changed: return False try: path.write_text(text, encoding="utf-8") except OSError: return False return True def default_model_id(family: str) -> str: """Derive a default server entry id from a family name.""" if family in PREFERRED_IDS: return PREFERRED_IDS[family] name = family if name.endswith("_tts"): name = name[:-4] return name.replace("_", "") or family def detect_audiocpp_dir() -> Optional[Path]: """Best-effort location of a local audio.cpp checkout with model_specs. Checks the AUDIOCPP_DIR environment variable, then ``app/audio.cpp`` in the tts-audiobook-generator root, then an ``audio.cpp`` directory in or above the current working directory. Returns the path only when it contains a ``model_specs`` directory. """ candidates: List[Path] = [] env_dir = os.environ.get("AUDIOCPP_DIR") if env_dir: candidates.append(Path(os.path.expanduser(env_dir))) candidates.append(APP_DIR / AUDIOCPP_DIR_NAME) cwd = Path.cwd() candidates.append(cwd / "audio.cpp") candidates.append(cwd.parent / "audio.cpp") candidates.append(cwd.parent.parent / "audio.cpp") for candidate in candidates: try: resolved = candidate.resolve() except OSError: continue if (resolved / "model_specs").is_dir(): return resolved return None # audio.cpp build directories are named ``--`` (e.g. # ``linux-cuda-release``, ``windows-vulkan-debug``, ``macos-metal-release``) # and the built server lands in ``/bin/audiocpp_server``. The Metal # macOS backend is reported as "cpu" here since it is not a separate # --backend choice for audiocpp_server. _BACKEND_TOKEN_RE = re.compile(r"-(cuda|vulkan|hip|cpu|metal)(?:-|$)") def detect_backend(audiocpp_dir: Path) -> Optional[str]: """Best-effort detection of the backend audiocpp_server was built for. Scans ``audiocpp_dir/build/*`` for build directories that contain a built ``bin/audiocpp_server`` (``.exe`` allowed on Windows) and reads the backend token out of the directory name (``-cuda-``, ``-vulkan-``, ``-hip-`` or ``-cpu-``; ``-metal-`` is mapped to ``cpu``). Returns the backend only when exactly one distinct backend was built, so a checkout with builds for several backends does not silently pick one. Returns None when there is no ``build/`` directory, no built server, or more than one distinct backend. """ build_root = audiocpp_dir / "build" if not build_root.is_dir(): return None backends: Set[str] = set() try: build_dirs = sorted(build_root.iterdir(), key=lambda p: p.name.lower()) except OSError: return None for build_dir in build_dirs: if not build_dir.is_dir(): continue server = build_dir / "bin" / "audiocpp_server" if not server.exists(): server_exe = build_dir / "bin" / "audiocpp_server.exe" if not server_exe.exists(): continue match = _BACKEND_TOKEN_RE.search(build_dir.name.lower()) if not match: continue token = match.group(1) backends.add("cpu" if token == "metal" else token) if len(backends) == 1: return next(iter(backends)) return None def _default_package(packages: List[dict]) -> Optional[dict]: """Pick the default package from a list of packages. Prefers the package flagged ``default: true``, then the first GGUF package, then the first package overall. Returns None for an empty list. """ if not packages: return None for package in packages: if package.get("default"): return package for package in packages: if package.get("format") == "gguf": return package return packages[0] def load_model_catalog(audiocpp_dir: Path) -> List[dict]: """Read model_specs/*.json and return the TTS-capable families. Each returned entry has: family, display_name, description, languages, clone_capable, packages (the full list from the spec), install_id (recommended package id), default_path (``models/``), and preferred_id. All families are treated equally and listed in alphabetical order by display name. """ specs_dir = audiocpp_dir / "model_specs" if not specs_dir.is_dir(): raise NotADirectoryError( f"{audiocpp_dir} has no model_specs/ directory; point " "--audiocpp-dir at an audio.cpp checkout") entries: List[dict] = [] for spec_path in sorted(specs_dir.glob("*.json")): try: spec = json.loads(spec_path.read_text(encoding="utf-8")) except (OSError, ValueError): continue tasks = spec.get("tasks") or [] if "tts" not in tasks and spec.get("category") != "tts": continue family = spec.get("family") or spec_path.stem packages = spec.get("packages") or [] package = _default_package(packages) if package is None: # No installable package: skip (cannot be hosted from a path). continue target_directory = package.get("target_directory") or family languages = spec.get("languages") or [] display_name = spec.get("display_name") or family description = spec.get("description") or "" entries.append({ "family": family, "display_name": display_name, "description": description, "languages": languages, "tasks": list(tasks), "clone_capable": "clone" in tasks, "packages": packages, "install_id": package.get("id") or family, "default_path": f"models/{target_directory}", "preferred_id": default_model_id(family), }) # All families are treated equally: alphabetical by display name. entries.sort(key=lambda entry: entry["display_name"].lower()) return entries def is_design_package(package: dict) -> bool: """Return True when a package's name marks it a voice-design model. audio.cpp voice-design packages (whose id, display name, or target directory mentions "voice design") are the only packages that must be hosted with task "vdes"; their role is not in the schema, only in those strings, so it is detected from them. """ text = " ".join(str(package.get(key, "")) for key in ("id", "display_name", "target_directory")) return bool(DESIGN_PACKAGE_RE.search(text)) def package_dir_options(entry: dict) -> List[dict]: """Return one option per distinct target_directory of a family's packages. Each option is a dict with: target_directory, install_id (the recommended package id inside that directory), design (voice-design package flag), and recommended (whether it holds the family's default package). Precisions that share a directory (q8_0/bf16/...) collapse to a single option. """ packages = entry.get("packages") or [] default_pkg = _default_package(packages) default_dir = (default_pkg or {}).get("target_directory") or entry["family"] by_dir: Dict[str, List[dict]] = {} order: List[str] = [] for package in packages: directory = package.get("target_directory") or entry["family"] if directory not in by_dir: by_dir[directory] = [] order.append(directory) by_dir[directory].append(package) options: List[dict] = [] for directory in order: package = _default_package(by_dir[directory]) options.append({ "target_directory": directory, "install_id": (package or {}).get("id") or directory, "design": is_design_package(package or {}), "recommended": directory == default_dir, }) # Put the recommended package first for a friendlier checklist. options.sort(key=lambda opt: not opt["recommended"]) return options def build_model_entry(family: str, model_id: str, model_path: str, task: str = TASK_TTS) -> dict: """Assemble one server.json model entry. ``task`` defaults to "tts"; voice design packages are hosted with "vdes" so the server runs its design session for speech requests (audiobook.py then requires --instructions with that entry). """ return { "id": model_id, "family": family, "path": model_path, "task": task, "mode": "offline", } def build_server_config(host: str, port: int, backend: str, lazy_load: bool, model_entries: List[dict], voice_dir: Optional[str] = None) -> dict: """Assemble the server.json document. ``voice_dir`` is a server-level cloning voice library; when set, every hosted clone-capable family can use its voices with ``--voice``. """ config_doc = { "host": host, "port": port, "backend": backend, "lazy_load": lazy_load, "models": model_entries, } if voice_dir: config_doc["voice_dir"] = voice_dir return config_doc def transcribe_wav_dir(wav_files: list, whisper_model: str) -> Dict[str, str]: """Transcribe each wav file and return a mapping of stem -> transcript.""" transcripts: Dict[str, str] = {} for wav_file in wav_files: name = wav_file.stem print(f"[INFO] Transcribing {wav_file.name} (voice '{name}')...") text = transcribe_reference_audio(str(wav_file), model_name=whisper_model) if text: print(f"[OK] {name}: {text}") else: print(f"[WARNING] No transcript for '{name}'; cloning works best " "with an accurate transcript — consider editing prompt_text " "by hand before starting the server") transcripts[name] = text or "" return transcripts def print_empty_transcript_warning(transcripts: Dict[str, str]) -> None: """Print a loud, final warning for voices whose transcript is empty.""" empty = sorted(name for name, text in transcripts.items() if not text) if not empty: return bar = "=" * 70 print() print(bar) print("[WARNING] MANUAL TRANSCRIPTION REQUIRED") print(bar) listing = " - " + "\n - ".join(empty) if len(empty) > 1 else f" - {empty[0]}" print(f"The following voice(s) have an EMPTY transcript in prompt_text:\n" f"{listing}") print("Those voices will NOT work until you add an accurate transcript.") print(f"Edit {PROMPT_TEXT_FILENAME} in your voice directory and fill in the " "text after '|' for each voice above.") print(bar) def _apply_port_sync(port: int, accepted: bool) -> None: """Write the port into app/converter/config.py, or report when declined.""" if accepted: if not update_config_api_url_port(port): print(f"[WARNING] Could not update {CONFIG_PATH}; edit " "AUDIOCPP_API_URL by hand so audiobook.py uses the " "new port") else: print("[WARNING] Left AUDIOCPP_API_URL unchanged; audiobook.py " f"will still use port {config_port()}") def _decide_transcription(wav_files: list, existing: Dict[str, str], prompt_exists: bool, force: bool, confirm: Callable[[str, bool], bool]) -> dict: """Decide which voices to transcribe; CONFIRM asks the plan questions. Returns a plan dict: {"mode": "all"|"missing"|"keep", "missing": [...], "existing": {...}} — "existing" carries the prompt_text mapping read while deciding, so the caller can reuse it instead of reading the file again. """ mode = "all" missing: List[Path] = [] if prompt_exists and not force: missing = [wav for wav in wav_files if not existing.get(wav.stem, "").strip()] if not missing: if confirm("All voices already transcribed in prompt_text. " "Re-transcribe anyway?", False): mode = "all" else: mode = "keep" elif confirm("Existing transcription and new .wavs detected, " "only transcribe new voices?", True): mode = "missing" else: mode = "all" return {"mode": mode, "missing": missing, "existing": existing} def _transcribe(args: argparse.Namespace, include_clone: bool, plan: dict) -> Tuple[Dict[str, str], bool]: """Transcribe the wav directory into a stem -> transcript mapping. Returns the mapping and a flag indicating whether it should be written to prompt_text (False when an existing, complete prompt_text is kept as-is). PLAN is always pre-collected — by the TUI (via _decide_transcription and its confirm callbacks) or by _flag_plan for a non-interactive run — so no questions are asked here. """ if not include_clone: print(f"[WARNING] Ignoring {args.input_dir}: no clone-capable family " "selected, so voice presets are not used") return {}, False wav_files = find_wav_files(args.input_dir) if not wav_files: print(f"[WARNING] No .wav files found in {args.input_dir}; writing the " "config without a voice_dir") return {}, False prompt_path = args.input_dir / PROMPT_TEXT_FILENAME existing = plan.get("existing") or {} if plan else {} if plan["mode"] == "keep": print(f"[INFO] Kept existing {prompt_path}; all voices were " "already transcribed, nothing new to transcribe") return existing, False if whisper_backend_available() is None: print("[WARNING] Neither faster_whisper nor whisper was found, so " "reference .wav files cannot be transcribed automatically and " "every transcript will be empty.") print(" Install whisper (or faster_whisper) in your " "audiobook environment to transcribe automatically; otherwise " "transcripts must be added by hand (see the warning at the end).") if plan["mode"] == "missing": new_transcripts = transcribe_wav_dir(plan["missing"], args.whisper_model) transcripts = dict(existing) transcripts.update(new_transcripts) else: transcripts = transcribe_wav_dir(wav_files, args.whisper_model) return transcripts, True def _flag_plan(wav_files: list, prompt_path: Path, force: bool) -> dict: """Build a transcription plan for a non-interactive (flag-only) run. With --force everything is re-transcribed; otherwise an existing prompt_text is reused and only voices with an empty transcript are re-transcribed, mirroring what the TUI confirms interactively. """ if prompt_path.exists() and not force: existing = read_prompt_text(prompt_path) missing = [wav for wav in wav_files if not existing.get(wav.stem, "").strip()] if not missing: return {"mode": "keep", "missing": [], "existing": existing} return {"mode": "missing", "missing": missing, "existing": existing} return {"mode": "all", "missing": [], "existing": {}} def _offer_config_model_id_sync(model_id: str, accepted: Optional[bool]) -> None: """Point app/converter/config.py at a single hosted model entry. The converter requests the model id configured in AUDIOCPP_MODEL_ID, and single-model servers use the same id for the clone entry, so both ids are rewritten together. ACCEPTED is True/False (apply/skip the rewrite) or None when no single-entry sync applies (nothing to do). """ if config.AUDIOCPP_MODEL_ID == model_id \ and config.AUDIOCPP_CLONE_MODEL_ID == model_id: return if accepted is None: return if accepted: if not update_config_model_ids(model_id, model_id): print(f"[WARNING] Could not update {CONFIG_PATH}; edit " "AUDIOCPP_MODEL_ID and AUDIOCPP_CLONE_MODEL_ID by hand so " "audiobook.py uses this model") else: print("[WARNING] Left the model ids unchanged; audiobook.py will " f"still request model '{config.AUDIOCPP_MODEL_ID}'") def _build_entries(family_keys: List[str], chosen: Dict[str, List[dict]], catalog_by_family: Dict[str, dict], task_picker: Callable[[str], str], id_picker: Callable[[str, str, str], str] ) -> Tuple[List[dict], List[str], List[Tuple[str, str]], List[str], bool]: """Build server.json model entries from the selected families/packages. TASK_PICKER is called for each design package to choose vdes/tts; ID_PICKER resolves a duplicate server entry id. Returns (model_entries, entry_ids, install_guidance, design_entry_ids, include_clone). """ model_entries: List[dict] = [] entry_ids: List[str] = [] install_guidance: List[Tuple[str, str]] = [] design_entry_ids: List[str] = [] include_clone = False for family in family_keys: entry = catalog_by_family[family] include_clone = include_clone or entry["clone_capable"] for opt in chosen[family]: task = task_picker(opt["install_id"]) if opt["design"] else TASK_TTS base_id = (f"{entry['preferred_id']}-design" if task == TASK_VDES else entry["preferred_id"]) model_id = base_id if model_id in entry_ids: model_id = id_picker(entry["display_name"], opt["install_id"], f"{base_id}-2") entry_ids.append(model_id) model_entries.append(build_model_entry( family, model_id, f"models/{opt['target_directory']}", task=task)) install_guidance.append((entry["display_name"], opt["install_id"])) if task == TASK_VDES: design_entry_ids.append(model_id) return (model_entries, entry_ids, install_guidance, design_entry_ids, include_clone) def _write_and_advise(audiocpp_dir: Path, wav_dir: Optional[Path], output_path: Path, model_entries: List[dict], install_guidance: List[Tuple[str, str]], host: str, port: int, backend: str, lazy_load: bool, transcripts: Dict[str, str], write_prompt: bool) -> None: """Console phase shared by both UI modes: write files, print summary. After a successful run the console output is the path of the written server.json. The model install commands (and optional automatic download) are handled separately by _install_models, called by both UI modes once the user has decided whether to download. """ voice_dir: Optional[str] = None if transcripts: if write_prompt: prompt_path = wav_dir / PROMPT_TEXT_FILENAME write_prompt_text(wav_dir, transcripts) print(f"[OK] Wrote {prompt_path}") voice_dir = str(wav_dir.resolve()) server_config = build_server_config( host=host, port=port, backend=backend, lazy_load=lazy_load, model_entries=model_entries, voice_dir=voice_dir) with output_path.open("w", encoding="utf-8") as handle: json.dump(server_config, handle, indent=2, ensure_ascii=False) handle.write("\n") count = len(model_entries) print(f"Wrote {output_path.resolve()} with {count} " f"{'entry' if count == 1 else 'entries'}.") def _install_models(audiocpp_dir: Path, install_guidance: List[Tuple[str, str]], download: bool) -> None: """Print and optionally run the model install commands. One ``python install `` command per hosted model (de-duped by install id). When DOWNLOAD is True each command is run in the audio.cpp checkout via ``subprocess.run`` so the models are downloaded automatically; a failing install is reported as a warning and does not abort the remaining downloads. When DOWNLOAD is False (or the model manager is missing) the commands are only printed, copy-pasteable as before. """ manager = audiocpp_dir / "tools" / "model_manager_v2.py" seen: Set[str] = set() install_ids: List[str] = [] for _, install_id in install_guidance: if install_id not in seen: seen.add(install_id) install_ids.append(install_id) if download and not manager.is_file(): print(f"[WARNING] {manager} not found; printing the install commands " "instead of running them") download = False for install_id in install_ids: command = f"python {manager} install {install_id}" if not download: print(command) continue print(f"[INFO] Downloading {install_id}...") try: result = subprocess.run( [sys.executable, str(manager), "install", install_id], cwd=str(audiocpp_dir)) except OSError as exc: print(f"[WARNING] Could not run {command}: {exc}") continue if result.returncode != 0: print(f"[WARNING] install {install_id} exited with code " f"{result.returncode}; the model may need to be downloaded " "by hand") def _decide_download(audiocpp_dir: Path, confirm: Callable[[str, bool], bool]) -> bool: """Ask whether to download the selected models now. CONFIRM asks the yes/no question (ask_bool for the line prompts, a TUI confirm for the wizard). When the audio.cpp model manager is missing the prompt is skipped and False is returned, so the install commands are only printed rather than offered to run. """ manager = audiocpp_dir / "tools" / "model_manager_v2.py" if not manager.is_file(): return False return confirm( "Automatically download the selected models with model_manager_v2.py " "now?", False) def _build_tree_families(catalog: List[dict]) -> List[dict]: """Shape the catalog into the checkbox_tree widget's family list.""" families: List[dict] = [] for entry in catalog: capabilities = ["tts"] if "clone" in entry["tasks"]: capabilities.append("cloning") if "design" in entry["tasks"]: capabilities.append("design") name = entry["display_name"] if name != entry["family"]: name = f"{name} ({entry['family']})" options = [] for opt in package_dir_options(entry): options.append({ "key": opt["target_directory"], "label": opt["install_id"], "recommended": opt["recommended"], }) families.append({ "label": name, "detail": ", ".join(capabilities), "options": options, }) return families def _wizard(stdscr, args: argparse.Namespace, parser: argparse.ArgumentParser ) -> Optional[dict]: """Run every TUI screen; return the collected settings, or None to abort. The wizard is a step state machine; each screen group is one step, and Esc anywhere but the first step falls back to the previous group (the widget returns the _GO_BACK sentinel, or a callback raises _GoBack). On the first screen (the audio.cpp checkout browser) Esc aborts the whole wizard as before. """ def ask_confirm(question: str, default: bool) -> bool: result = tui.confirm(stdscr, question, default=default, cancel_value=_GO_BACK) if result is _GO_BACK: raise _GoBack() return result step = 0 while True: if step == 0: # Checkout browser + the output path/overwrite confirmation. The # browser asks for the checkout root and finds model_specs/ inside # it (picking the model_specs directory itself works too — its # parent is used). A highlighted subdirectory named "audio.cpp" # that already contains model_specs/ is auto-accepted on # Enter/Right, skipping the "[ Use this directory ]" step. # Pressing Esc on an overwrite confirmation returns here instead # of aborting: the browser then restarts inside the previously # accepted checkout with auto-accept disabled, so a wrong guess # can be corrected. An explicit --audiocpp-dir flag has no # browser to return to, so Esc still aborts there. Esc on the # browser itself is the first step, so it aborts the wizard. auto_accept = True browser_start: Path = Path.cwd() force_browse = False def do_browse(): return tui.browse_directory( stdscr, "Select your audio.cpp directory", validate=lambda p: None if _resolve_audiocpp_root(p) else "No model_specs/ directory here", info=_audiocpp_root_status, preview=_audiocpp_root_preview, help_lines=["The root folder of your audio.cpp " "checkout;", "it is the one that contains " "model_specs/"], start=browser_start, auto_select=_checkout_auto_select if auto_accept else None) while True: audiocpp_dir = args.audiocpp_dir if audiocpp_dir is None and not force_browse: audiocpp_dir = find_local_checkout() if force_browse: audiocpp_dir = None if audiocpp_dir is None: if force_browse: # Esc on an overwrite confirmation came back here: go # straight back into the browser inside the previously # accepted checkout (auto-accept disabled). audiocpp_dir = do_browse() else: # No checkout found anywhere: offer to clone one into # ./app/audio.cpp or browse for an existing checkout. # Esc on this first menu aborts the wizard. choice = tui.menu( stdscr, "No audio.cpp checkout found", [(f"Clone into ./app/{AUDIOCPP_DIR_NAME} " f"(from {AUDIOCPP_GIT_URL})", "clone"), ("Browse for an existing checkout", "browse")], help_lines=[ "audio.cpp hosts the TTS model families " "this generator uses.", "Clone it into the project's app " "directory, or point at an existing " "checkout."]) if choice == "clone": target = APP_DIR / AUDIOCPP_DIR_NAME with tui.suspend(stdscr): rc = common.git_clone(AUDIOCPP_GIT_URL, target) if rc != 0: raise _TuiError( f"git clone failed (exit {rc}). Clone " f"audio.cpp manually: git clone " f"{AUDIOCPP_GIT_URL} {target}") audiocpp_dir = target else: audiocpp_dir = do_browse() audiocpp_dir = Path(audiocpp_dir).resolve() if not audiocpp_dir.is_dir(): raise _TuiError(f"audio.cpp checkout not found: " f"{audiocpp_dir}") root = _resolve_audiocpp_root(audiocpp_dir) if root is None: raise _TuiError( f"{audiocpp_dir} has no model_specs/ directory; " "select the root of your audio.cpp checkout") audiocpp_dir = root try: catalog = load_model_catalog(audiocpp_dir) except NotADirectoryError as exc: raise _TuiError(str(exc)) if not catalog: raise _TuiError(f"No TTS model families found in " f"{audiocpp_dir}/model_specs; check the " "checkout is up to date") catalog_by_family = {entry["family"]: entry for entry in catalog} output_path = args.output if args.output is not None \ else audiocpp_dir / "server.json" esc_back = args.audiocpp_dir is None went_back = False if not args.force and output_path.exists(): decision = tui.confirm( stdscr, f"{output_path} already exists. Overwrite?", default=True, cancel_value=_GO_BACK if esc_back else None) if decision is _GO_BACK: went_back = True elif decision is False: if args.output is None: output_path = Path.cwd() / "server.json" if output_path.exists(): decision = tui.confirm( stdscr, f"{output_path} already exists. " "Overwrite?", default=True, cancel_value=_GO_BACK if esc_back else None) if decision is _GO_BACK: went_back = True elif decision is False: return None else: return None if went_back: auto_accept = False browser_start = audiocpp_dir force_browse = True continue break detected_backend = detect_backend(audiocpp_dir) step = 1 continue if step == 1: # Families and packages (flag or tree). Esc returns to the # checkout browser (step 0). chosen: Dict[str, List[dict]] = {} if args.families is not None: requested = [f.strip() for f in args.families.split(",") if f.strip()] unknown = [f for f in requested if f not in catalog_by_family] if unknown: raise _TuiError( f"Unknown family in --families: {', '.join(unknown)}. " f"Available: {', '.join(catalog_by_family)}") family_keys: List[str] = [] for family in requested: if family not in family_keys: family_keys.append(family) chosen[family] = [opt for opt in package_dir_options( catalog_by_family[family]) if opt["recommended"]] else: tree_families = _build_tree_families(catalog) picked = tui.checkbox_tree( stdscr, "Select TTS model families to host", tree_families, expand_all=args.all_packages, back_value=_GO_BACK) if picked is _GO_BACK: step = 0 continue family_keys = [] for family_index, option_key in picked: family = catalog[family_index]["family"] if family not in chosen: chosen[family] = [] family_keys.append(family) chosen[family].append(option_key) for family in list(chosen): keyed = {opt["target_directory"]: opt for opt in package_dir_options( catalog_by_family[family])} chosen[family] = [keyed[key] for key in chosen[family]] step = 2 continue if step == 2: # Design task menus and duplicate-id renames. Esc anywhere here # falls back to the families tree (step 1). def task_picker(install_id: str) -> str: result = tui.menu( stdscr, f"How should the '{install_id}' package be hosted?", [ ("design (vdes) - describe the voice with " "--instructions", TASK_VDES), ("tts - normal synthesis", TASK_TTS), ], default_index=0, back_value=_GO_BACK) if result is _GO_BACK: raise _GoBack() return result def id_picker(display_name: str, install_id: str, default: str) -> str: result = tui.line_edit( stdscr, f"Server model id for {display_name} package " f"'{install_id}'", default, back_value=_GO_BACK) if result is _GO_BACK: raise _GoBack() return result try: model_entries, entry_ids, install_guidance, \ design_entry_ids, include_clone = _build_entries( family_keys, chosen, catalog_by_family, task_picker, id_picker) except _GoBack: step = 1 continue step = 3 continue if step == 3: # Server settings (host, port, port-sync, backend, lazy). Esc on # any of them falls back to the previous group (step 2). if args.host: host = args.host else: host = tui.line_edit( stdscr, "Bind host", DEFAULT_HOST, help_lines=["The IP address audiocpp will be hosted on", "127.0.0.1 (this machine) is probably " "correct"], back_value=_GO_BACK) if host is _GO_BACK: step = 2 continue if args.port is not None: port = args.port else: port_text = tui.line_edit( stdscr, "Port", str(config_port()), validate=lambda s: None if (s.isdigit() and 1 <= int(s) <= 65535) else "Enter a port number between 1 and 65535", help_lines=["The port audiocpp will be hosted on"], back_value=_GO_BACK) if port_text is _GO_BACK: step = 2 continue port = int(port_text) sync_port: Optional[bool] = None if port != config_port(): sync_port = tui.confirm( stdscr, f"Update AUDIOCPP_API_URL in app/converter/config.py " f"to port {port} so audiobook.py talks to this server", default=True, cancel_value=_GO_BACK) if sync_port is _GO_BACK: step = 2 continue if args.build_backend: backend = args.build_backend build = detected_backend is None elif args.backend: backend = args.backend build = False elif detected_backend is not None: # Already built: use the detected backend, no menu, no build. backend = detected_backend build = False else: backend_options, backend_default = _backend_options(None) backend = tui.menu( stdscr, "Which inference backend was audiocpp_server " "built for?", backend_options, default_index=backend_default, back_value=_GO_BACK) if backend is _GO_BACK: step = 2 continue # Not built for any backend yet: offer to build it now. The # build itself runs in the console tail after the wizard. build = tui.confirm( stdscr, f"audiocpp_server is not built for {backend}. " f"Build it now (runs scripts/build_*)?", default=True, cancel_value=_GO_BACK) if build is _GO_BACK: step = 2 continue default_lazy = len(model_entries) > 1 if args.lazy_load: lazy_load = True else: lazy_load = tui.confirm( stdscr, "Load models lazily (on first use instead of at " "startup)", default=default_lazy, cancel_value=_GO_BACK) if lazy_load is _GO_BACK: step = 2 continue step = 4 continue if step == 4: # Wav directory (flag, browsed when cloning, else skipped). Esc # falls back to the server settings (step 3). if args.input_dir is not None: wav_dir = args.input_dir elif include_clone: wav_start = detect_wav_dir(audiocpp_dir, TTS_ROOT) wav_dir = tui.browse_directory( stdscr, "Select the directory with your .wav voices", info=_wav_dir_info, preview=_wav_dir_preview, start=wav_start if wav_start is not None else VOICES_DIR, back_value=_GO_BACK) if wav_dir is _GO_BACK: step = 3 continue else: wav_dir = None step = 5 continue if step == 5: # Transcription plan (questions only; transcription runs after). # Esc falls back to the wav browser (step 4). plan: Optional[dict] = None if include_clone and wav_dir is not None: wav_files = find_wav_files(wav_dir) if wav_files: prompt_path = wav_dir / PROMPT_TEXT_FILENAME existing = read_prompt_text(prompt_path) if ( prompt_path.exists() and not args.force) else {} try: plan = _decide_transcription( wav_files, existing, prompt_path.exists(), args.force, ask_confirm) except _GoBack: step = 4 continue step = 6 continue if step == 6: # Single-model id sync decision. Esc falls back to the # transcription plan (step 5). sync_model_ids: Optional[bool] = None if len(entry_ids) == 1 and not ( config.AUDIOCPP_MODEL_ID == entry_ids[0] and config.AUDIOCPP_CLONE_MODEL_ID == entry_ids[0]): sync_model_ids = tui.confirm( stdscr, "Update AUDIOCPP_MODEL_ID and " "AUDIOCPP_CLONE_MODEL_ID in app/converter/config.py to " f"'{entry_ids[0]}' so audiobook.py uses this model", default=True, cancel_value=_GO_BACK) if sync_model_ids is _GO_BACK: step = 5 continue step = 8 continue if step == 8: # Automatic model download (or print the install commands). Esc # falls back to the model-id sync (step 6). try: download = _decide_download(audiocpp_dir, ask_confirm) except _GoBack: step = 6 continue return { "audiocpp_dir": audiocpp_dir, "catalog": catalog, "catalog_by_family": catalog_by_family, "output_path": output_path, "family_keys": family_keys, "chosen": chosen, "model_entries": model_entries, "entry_ids": entry_ids, "install_guidance": install_guidance, "design_entry_ids": design_entry_ids, "include_clone": include_clone, "host": host, "port": port, "backend": backend, "build": build, "lazy_load": lazy_load, "sync_port": sync_port, "sync_model_ids": sync_model_ids, "wav_dir": wav_dir, "plan": plan, "download": download, } def find_local_checkout() -> Optional[Path]: """Best-effort location of an audio.cpp checkout with model_specs. Checks the AUDIOCPP_DIR environment variable, then ``app/audio.cpp`` inside the tts-audiobook-generator root, then an ``audio.cpp`` directory in or above the current working directory. Returns the path only when it contains a ``model_specs`` directory. """ candidates: List[Path] = [] env_dir = os.environ.get("AUDIOCPP_DIR") if env_dir: candidates.append(Path(os.path.expanduser(env_dir))) candidates.append(APP_DIR / AUDIOCPP_DIR_NAME) cwd = Path.cwd() candidates.append(cwd / AUDIOCPP_DIR_NAME) candidates.append(cwd.parent / AUDIOCPP_DIR_NAME) candidates.append(cwd.parent.parent / AUDIOCPP_DIR_NAME) for candidate in candidates: try: resolved = candidate.resolve() except OSError: continue if (resolved / "model_specs").is_dir(): return resolved return None def fetch_server_models(api_url: str) -> Optional[List[Dict[str, str]]]: """List a running audiocpp_server's model entries via GET /v1/models. Returns ``[{id, family, task}, ...]`` — the same shape the converter's client resolves at startup — or None when URL does not answer with a valid document (wrong server, still starting, older audio.cpp). Used by the hub to drive the convert menus against a remote server that has no local server.json describing it. """ try: with urllib.request.urlopen( f"{api_url.rstrip('/')}/v1/models", timeout=10) as response: payload = json.loads(response.read().decode("utf-8")) except (OSError, ValueError): # URLError/HTTPError/socket errors are OSErrors; a non-JSON body is # a ValueError. Anything else means "not an audiocpp_server". return None entries = payload.get("data") if isinstance(payload, dict) else None models: List[Dict[str, str]] = [] for entry in entries or []: if isinstance(entry, dict) and entry.get("id"): models.append({ "id": str(entry["id"]), "family": str(entry.get("family") or ""), "task": str(entry.get("task") or ""), }) return models def fetch_server_voices(api_url: str, model_id: str) -> Optional[List[str]]: """List a running audiocpp_server's voices for MODEL_ID. Queries ``GET /v1/audio/voices?model=`` — the endpoint the converter validates ``--voice`` against — and returns its voice-name list, or None when the server cannot be queried. Lets the hub offer a remote server's voices without reading its configuration locally. """ query = urllib.parse.urlencode({"model": model_id}) try: with urllib.request.urlopen( f"{api_url.rstrip('/')}/v1/audio/voices?{query}", timeout=10) as response: payload = json.loads(response.read().decode("utf-8")) except (OSError, ValueError): return None voices = payload.get("voices") if isinstance(payload, dict) else None if not isinstance(voices, list): return None return [str(voice) for voice in voices] def find_audiocpp_server_bin(audiocpp_dir: Path) -> Optional[Path]: """Return the built audiocpp_server binary, or None when not built. Scans ``audiocpp_dir/build/*`` for a build directory containing ``bin/audiocpp_server`` (``.exe`` allowed on Windows). When several builds exist the first (alphabetical) is returned. """ build_root = audiocpp_dir / "build" if not build_root.is_dir(): return None try: build_dirs = sorted(build_root.iterdir(), key=lambda p: p.name.lower()) except OSError: return None for build_dir in build_dirs: if not build_dir.is_dir(): continue for name in ("audiocpp_server", "audiocpp_server.exe"): server = build_dir / "bin" / name if server.exists(): return server return None def find_build_script(audiocpp_dir: Path) -> Optional[Path]: """Return the audio.cpp build helper script to run, or None. Prefers ``scripts/build_linux.sh``; otherwise the first ``scripts/build_*.sh`` it finds. (Windows ``.bat`` scripts are not run automatically — build manually there.) """ scripts = audiocpp_dir / "scripts" if not scripts.is_dir(): return None preferred = scripts / "build_linux.sh" if preferred.exists(): return preferred try: candidates = sorted(scripts.glob("build_*.sh"), key=lambda p: p.name.lower()) except OSError: return None return candidates[0] if candidates else None def build_audiocpp(audiocpp_dir: Path, backend: str) -> int: """Build audiocpp_server for BACKEND, streaming output to the console. Returns the build script's exit code (non-zero when the script is missing). Run from a console context (after the TUI wizard returns, or inside ``tui.suspend``). """ script = find_build_script(audiocpp_dir) if script is None: print(f"[ERROR] No build script found in {audiocpp_dir}/scripts; " "build audiocpp_server manually (see the audio.cpp README)") return 1 print(f"[INFO] Building audiocpp_server for {backend} " f"({script} --backend {backend} --target audiocpp_server)...") return common.run_console_subprocess( ["sh", str(script), "--backend", backend, "--target", "audiocpp_server"], cwd=audiocpp_dir) def _print_launch_hint(audiocpp_dir: Path, output_path: Path) -> None: """Print the exact command to start the server (or build guidance).""" binary = find_audiocpp_server_bin(audiocpp_dir) print() if binary is not None: print("Start the server with:") print(f" {binary} --config {output_path}") else: print("[INFO] audiocpp_server binary not found. Build it first, e.g.:") script = find_build_script(audiocpp_dir) if script is not None: print(f" sh {script} --backend " "--target audiocpp_server") print(f" then run: ./build/--release/bin/" f"audiocpp_server --config {output_path}") def _execute(settings: dict, args: argparse.Namespace) -> int: """Shared console tail: build, sync, transcribe, write, install, advise. Runs after the TUI wizard returns (or after _collect_from_flags for a non-interactive run): the terminal is plain, so subprocess output and transcription progress appear normally. """ audiocpp_dir = settings["audiocpp_dir"] # Build audiocpp_server first (the longest step), when requested. if settings.get("build"): rc = build_audiocpp(audiocpp_dir, settings["backend"]) if rc != 0: print(f"[WARNING] build exited with code {rc}; the server.json " "was still written — build audiocpp_server manually before " "starting it") else: print("[OK] build complete") # Port sync (applied now that the terminal is back). if settings["sync_port"] is True: _apply_port_sync(settings["port"], True) elif settings["sync_port"] is False: _apply_port_sync(settings["port"], False) # Transcription (console; the questions were already answered). args.input_dir = settings["wav_dir"] if settings["include_clone"] and args.input_dir is not None: transcripts, write_prompt = _transcribe(args, True, plan=settings["plan"]) elif args.input_dir is not None: print(f"[WARNING] Ignoring {args.input_dir}: no clone-capable family " "selected, so voice presets are not used") transcripts, write_prompt = {}, False else: transcripts, write_prompt = {}, False _write_and_advise( audiocpp_dir, settings["wav_dir"], settings["output_path"], settings["model_entries"], settings["install_guidance"], settings["host"], settings["port"], settings["backend"], settings["lazy_load"], transcripts, write_prompt) if len(settings["entry_ids"]) == 1: _offer_config_model_id_sync(settings["entry_ids"][0], settings["sync_model_ids"]) print_empty_transcript_warning(transcripts) _install_models(audiocpp_dir, settings["install_guidance"], settings["download"]) _print_launch_hint(audiocpp_dir, settings["output_path"]) return 0 def run_tui(args: Optional[argparse.Namespace] = None, parser: Optional[argparse.ArgumentParser] = None) -> int: """Run the audio.cpp setup wizard end-to-end. With no ARGS (the hub's call) a default namespace is built so the full wizard runs. Called from ``main`` after argparse when the terminal is interactive. Returns the process exit code. """ import curses if args is None: parser = build_parser() args = parser.parse_args([]) if args.input_dir is not None and not args.input_dir.is_dir(): print(f"[ERROR] --wavs not found: {args.input_dir}", file=sys.stderr) return 2 try: settings = curses.wrapper(_wizard, args, parser) except _TuiError as exc: print(f"[ERROR] {exc}", file=sys.stderr) return 2 except tui.WizardCancelled: print("\n[INFO] Cancelled; nothing was written") return 1 try: curses.curs_set(1) # restore the text cursor hidden by the TUI except curses.error: pass if settings is None: print("[INFO] Aborted; existing server.json kept") return 1 return _execute(settings, args) def _collect_from_flags(args: argparse.Namespace, parser: argparse.ArgumentParser) -> Optional[dict]: """Build the settings dict from flags for a non-interactive run. Every required value must come from a flag (there are no prompts in a non-interactive run); a missing one is a hard ``parser.error``. Returns the settings dict, or None when the user declined an overwrite (the default-location fallback then also exists). """ # Checkout: --audiocpp-dir, else a local checkout, else --clone clones one. audiocpp_dir = args.audiocpp_dir if audiocpp_dir is None: audiocpp_dir = find_local_checkout() if audiocpp_dir is None and args.clone: target = APP_DIR / AUDIOCPP_DIR_NAME rc = common.git_clone(AUDIOCPP_GIT_URL, target) if rc != 0: parser.error(f"git clone failed (exit {rc}); clone audio.cpp " f"manually: git clone {AUDIOCPP_GIT_URL} {target}") audiocpp_dir = target if audiocpp_dir is None: parser.error( "An audio.cpp checkout is required. Pass --audiocpp-dir PATH, " "or --clone to clone app/audio.cpp, or run without flags for the " "TUI wizard.") audiocpp_dir = Path(audiocpp_dir).resolve() if not audiocpp_dir.is_dir(): parser.error(f"audio.cpp checkout not found: {audiocpp_dir}") root = _resolve_audiocpp_root(audiocpp_dir) if root is None: parser.error(f"{audiocpp_dir} has no model_specs/ directory; point " "--audiocpp-dir at the root of an audio.cpp checkout") audiocpp_dir = root try: catalog = load_model_catalog(audiocpp_dir) except NotADirectoryError as exc: parser.error(str(exc)) if not catalog: parser.error( f"No TTS model families found in {audiocpp_dir}/model_specs; " "check the checkout is up to date") catalog_by_family = {entry["family"]: entry for entry in catalog} # Families: required from --families in a non-interactive run. if args.families is None: parser.error("--families is required in a non-interactive run (or run " "without flags for the TUI wizard)") requested = [f.strip() for f in args.families.split(",") if f.strip()] unknown = [f for f in requested if f not in catalog_by_family] if unknown: parser.error( f"Unknown family in --families: {', '.join(unknown)}. " f"Available: {', '.join(catalog_by_family)}") family_keys: List[str] = [] for fam in requested: if fam not in family_keys: family_keys.append(fam) chosen: Dict[str, List[dict]] = {} for family in family_keys: opts = package_dir_options(catalog_by_family[family]) if args.all_packages: chosen[family] = opts else: chosen[family] = [opt for opt in opts if opt["recommended"]] # Non-interactive pickers: design packages default to vdes, dup ids get -2. def task_picker(install_id: str) -> str: return TASK_VDES def id_picker(display_name: str, install_id: str, default: str) -> str: return default model_entries, entry_ids, install_guidance, design_entry_ids, include_clone = \ _build_entries(family_keys, chosen, catalog_by_family, task_picker, id_picker) # Server settings. host = args.host or DEFAULT_HOST detected_backend = detect_backend(audiocpp_dir) if args.build_backend: backend = args.build_backend build = detected_backend is None elif args.backend: backend = args.backend build = False elif detected_backend is not None: backend = detected_backend build = False else: backend = "cuda" build = False port = args.port if args.port is not None else config_port() lazy_load = args.lazy_load if args.lazy_load else (len(model_entries) > 1) # Output path / overwrite (decline falls back to cwd, then aborts). output_path = args.output if args.output is not None \ else audiocpp_dir / "server.json" if output_path.exists() and not args.force: if args.output is None: output_path = Path.cwd() / "server.json" if output_path.exists() and not args.force: print("[INFO] Aborted; existing server.json kept") return None else: print("[INFO] Aborted; existing server.json kept") return None # Config sync decisions (auto-apply unless explicitly declined). sync_port: Optional[bool] = None if port != config_port(): sync_port = not args.no_sync_port sync_model_ids: Optional[bool] = None if len(entry_ids) == 1 and not ( config.AUDIOCPP_MODEL_ID == entry_ids[0] and config.AUDIOCPP_CLONE_MODEL_ID == entry_ids[0]): sync_model_ids = not args.no_sync_model_ids # Wav dir + transcription plan (defaults to the project's voices/ dir). wav_dir = args.input_dir if args.input_dir is not None else VOICES_DIR plan: Optional[dict] = None if include_clone and wav_dir is not None: wav_files = find_wav_files(wav_dir) if wav_files: prompt_path = wav_dir / PROMPT_TEXT_FILENAME plan = _flag_plan(wav_files, prompt_path, args.force) return { "audiocpp_dir": audiocpp_dir, "catalog": catalog, "catalog_by_family": catalog_by_family, "output_path": output_path, "family_keys": family_keys, "chosen": chosen, "model_entries": model_entries, "entry_ids": entry_ids, "install_guidance": install_guidance, "design_entry_ids": design_entry_ids, "include_clone": include_clone, "host": host, "port": port, "backend": backend, "build": build, "lazy_load": lazy_load, "sync_port": sync_port, "sync_model_ids": sync_model_ids, "wav_dir": wav_dir, "plan": plan, "download": args.download, } def build_parser() -> argparse.ArgumentParser: """The audio.cpp setup CLI (also used to build a default namespace).""" parser = argparse.ArgumentParser( description="Set up the audio.cpp TTS backend: clone/build, pick " "models, write server.json, and sync app/converter/config.py.") parser.add_argument("--wavs", type=resolve_wav_dir_arg, default=None, dest="input_dir", metavar="WAV_DIR", help="Directory with .wav reference files to publish as " "a server-level voice_dir cloning library " f"(default: {VOICES_DIR}; asked for when omitted " "in the TUI)") parser.add_argument("--output", type=Path, default=None, help="Output path for server.json (default: " "server.json inside the audio.cpp checkout; an " "existing file is overwritten only with --force " "or a TUI confirm)") parser.add_argument("--audiocpp-dir", type=normalize_dir_arg, default=None, help="Path to a local audio.cpp checkout containing a " "model_specs/ directory (default: detected from " "AUDIOCPP_DIR or ./app/audio.cpp; in the TUI you can " "clone one instead)") parser.add_argument("--clone", action="store_true", help="Non-interactive: clone audio.cpp into " "./app/audio.cpp when no checkout is found") parser.add_argument("--families", type=str, default=None, help="Comma-separated model families to host, as named " "in the audio.cpp catalog (e.g. " "qwen3_tts,higgs_audio_tts). Required in a " "non-interactive run; skips the family tree in " "the TUI") parser.add_argument("--all-packages", action="store_true", help="Host every installable package of each selected " "family (distinct target_directory) instead of " "only the recommended one. Voice-design packages " "are hosted with task 'vdes'") parser.add_argument("--host", type=str, default=None, help="Bind host for the server (default: 127.0.0.1)") parser.add_argument("--port", type=int, default=None, help="Port for the server (default: the port in " "AUDIOCPP_API_URL from app/converter/config.py)") parser.add_argument("--backend", choices=BACKENDS, default=None, help="Inference backend recorded in server.json " "(default: auto-detected from the checkout's " "build/ directory, else cuda)") parser.add_argument("--build-backend", choices=BACKENDS, default=None, help="Build audiocpp_server for this backend when it " "is not built yet, and use it in server.json") parser.add_argument("--lazy-load", action="store_true", help="Load models on first use instead of at startup " "(default: on when more than one model is hosted)") parser.add_argument("--whisper-model", type=str, default="base", help="Whisper model size for transcription " "(default: base)") parser.add_argument("--force", action="store_true", help="Overwrite the output file (and prompt_text) " "without prompting") parser.add_argument("--download", action="store_true", help="Run model_manager_v2.py install for each hosted " "model automatically (default: print the commands " "only)") parser.add_argument("--no-sync-port", action="store_true", help="Do not rewrite AUDIOCPP_API_URL in " "app/converter/config.py when --port differs") parser.add_argument("--no-sync-model-ids", action="store_true", help="Do not rewrite AUDIOCPP_MODEL_ID/" "AUDIOCPP_CLONE_MODEL_ID for a single-entry server") return parser def detect() -> BackendStatus: """Detect how far audio.cpp is set up, plus the command to start it.""" checkout = find_local_checkout() # Probe the server first: it may be running externally even with no # local checkout, and the status table should show that. running = common.server_running(config.AUDIOCPP_API_URL) details: List[str] = [] launch = "" if checkout is None: return BackendStatus("audiocpp", "audio.cpp", installed=False, configured=False, running=running, details=["not cloned — run setup to clone " "./app/audio.cpp"]) details.append(f"checkout: {checkout}") binary = find_audiocpp_server_bin(checkout) built = binary is not None if built: details.append(f"built: {binary}") else: details.append("not built — run setup to build audiocpp_server") server_json = checkout / "server.json" configured = server_json.exists() specs: List[ServerSpec] = [] if configured: details.append(f"config: {server_json}") if built: specs = [ServerSpec( "audiocpp", config.AUDIOCPP_API_URL, [str(binary), "--config", str(server_json)])] else: launch = (f"./build/--release/bin/" f"audiocpp_server --config {server_json}") else: details.append("no server.json — run setup to configure models") if specs: launch = format_launch_hint(specs) return BackendStatus("audiocpp", "audio.cpp", installed=built, configured=configured, running=running, details=details, launch_hint=launch, servers=specs, managed=servers.manages(specs)) configure_actions: List[ConfigureAction] = [ ConfigureAction("Reconfigure audio.cpp (models, voices, server.json)", run_tui), ] def main() -> int: parser = build_parser() args = parser.parse_args() if args.input_dir is not None and not args.input_dir.is_dir(): parser.error( f"WAV directory not found: {args.input_dir}\n" f" (resolved from the current working directory: " f"{Path.cwd()})\n" " --wavs must be a directory containing the .wav " "reference files to use as voice cloning presets") if _interactive(): return run_tui(args, parser) # Non-interactive (no terminal, or all flags supplied): flag-only path. settings = _collect_from_flags(args, parser) if settings is None: return 1 return _execute(settings, args) if __name__ == "__main__": sys.exit(main())