#!/usr/bin/env python3 """Set up the audio.cpp TTS backend for the audiobook generator. This does the whole audio.cpp setup end-to-end as a full-screen DOS-style TUI: locate or clone an audio.cpp checkout into ``app/audio.cpp``, optionally build ``audiocpp_server``, pick model families/packages from the checkout's ``model_specs`` catalog, transcribe reference .wav voices, write ``server.json``, sync ``app/converter/config.py``, download the models, and print the exact command to start the server. It is driven by ``audiobook.py``'s TUI hub (``backends.REGISTRY``) but can also be run directly for scripting — every value has a flag, and a non-interactive run with all flags supplied never opens the TUI. The converter is family-agnostic (it detects the family of the selected entry from ``GET /v1/models`` at startup), so any TTS family listed in the catalog works without further changes. Usage: python app/backends/audiocpp.py [--wavs WAV_DIR] [--output PATH] [--audiocpp-dir PATH] [--clone] [--families FAM1,FAM2] [--all-packages] [--host HOST] [--port PORT] [--build-backend {cuda,vulkan,hip,cpu}] [--backend {cuda,vulkan,hip,cpu}] [--whisper-model NAME] [--force] [--download] [--no-sync-port] [--no-sync-model-ids] With no flags and a terminal, the TUI wizard runs. Without a terminal (or with all flags supplied), it runs non-interactively from the flags; any missing required value is a hard error with a remediation hint. When the target ``server.json`` already exists, the TUI wizard runs as a "modify": it loads the existing models, host, port, backend and voice directory and pre-fills the screens with them (the model tree opens with the installed models already checked) instead of prompting to overwrite, and offers to delete already-downloaded models that are no longer selected. """ import argparse import json import os import re import shlex import shutil import sys import tempfile import urllib.parse import urllib.request from datetime import datetime from pathlib import Path from typing import Callable, Dict, List, Optional, Set, Tuple # Allow running directly (python app/backends/audiocpp.py) from any cwd. sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) from backends import ( BackendStatus, ServerSpec, common, format_launch_hint, probe, servers, ) from backends.common import ( APP_DIR, CONFIG_PATH, PROMPT_TEXT_FILENAME, TTS_ROOT, VOICES_DIR, detect_wav_dir, find_wav_files, normalize_dir_arg, read_prompt_text, resolve_wav_dir_arg, write_prompt_text, ) from backends.common import ( wav_dir_info as _wav_dir_info, ) from backends.common import ( wav_dir_preview as _wav_dir_preview, ) from converter import config from converter.tts import transcribe_reference_audio, whisper_backend_available from ui import taskview, tui DEFAULT_HOST = "127.0.0.1" FALLBACK_PORT = 8080 BACKENDS = ("cuda", "vulkan", "hip", "cpu") TASK_TTS = "tts" TASK_VDES = "vdes" # audio.cpp is cloned into a sibling directory of the audiobook generator. AUDIOCPP_DIR_NAME = "audio.cpp" AUDIOCPP_GIT_URL = "https://github.com/0xShug0/audio.cpp" # Sentinel returned by tui.confirm (via its cancel_value) when the user # presses Esc on an overwrite prompt to go back to the wav-directory browser # instead of aborting the wizard. _GO_BACK = object() class _GoBack(Exception): """Internal signal: Esc was pressed inside one of a screen's sub-prompts. The wizard drives a stack of screens via ``tui.Wizard``. Helpers that ask several questions through callbacks (the task/id pickers inside ``_build_entries``, the transcription plan, the download prompt) cannot themselves return the wizard's ``BACK`` sentinel, so they convert the ``_GO_BACK`` value passed to each widget into this exception. The screen that invoked the helper catches it and returns ``tui.Wizard.BACK``, which pops back to the previous screen. Esc on the first screen aborts the whole wizard. """ # Package names that mark a voice-design model (hosted with task "vdes"). DESIGN_PACKAGE_RE = re.compile(r"voice[\s_\-]?design", re.IGNORECASE) # Short, friendly default entry ids for selected families. Other families # derive an id from their family name (see default_model_id). All families # are listed equally, in alphabetical order. PREFERRED_IDS = { "qwen3_tts": "qwen", "higgs_audio_tts": "higgs", "voxcpm2": "voxcpm2", "index_tts2": "indextts2", } class _TuiError(Exception): """A fatal error raised from inside the TUI wizard. The message is reported to stderr after the terminal is restored; the process exits with code 2 (matching a parser error). """ def _interactive() -> bool: """True when the TUI wizard can run (curses importable + tty).""" try: import curses # noqa: F401 except ImportError: return False try: return sys.stdin.isatty() and sys.stdout.isatty() except (AttributeError, ValueError): return False def _resolve_audiocpp_root(directory: Path) -> Optional[Path]: """Return the audio.cpp checkout root for DIRECTORY, or None. Accepts either the checkout root itself (it must contain a ``model_specs`` directory) or the ``model_specs`` directory inside it (the parent is used), so the file browser cannot pick the wrong one of the two. """ if (directory / "model_specs").is_dir(): return directory if directory.name == "model_specs" and directory.is_dir(): return directory.parent return None # Backend display order, with short descriptions. The backend name is padded # so the descriptions' dashes line up in the menu. _BACKEND_DESCRIPTIONS = ( ("cuda", "NVIDIA GPUs (fastest)"), ("vulkan", "cross-vendor GPU"), ("hip", "AMD GPUs"), ("cpu", "no GPU required"), ) def _backend_options(detected: Optional[str] = None ) -> Tuple[List[Tuple[str, str]], int]: """Build the aligned backend menu options and the default index. The backend names are padded to a common width so the ``-`` dashes before the descriptions line up. When DETECTED matches one of the options, that option gets ``[auto-detected]`` appended and is the default (cursor/start) selection; otherwise the first option is the default as before. Returns (options, default_index). """ width = max(len(name) for name, _ in _BACKEND_DESCRIPTIONS) options: List[Tuple[str, str]] = [] default_index = 0 for index, (name, desc) in enumerate(_BACKEND_DESCRIPTIONS): label = f"{name.ljust(width)} - {desc}" if detected == name: label += " [auto-detected]" default_index = index options.append((label, name)) return options, default_index def config_port() -> int: """Return the port of AUDIOCPP_API_URL in app/converter/config.py.""" try: return urllib.parse.urlsplit(config.AUDIOCPP_API_URL).port or FALLBACK_PORT except ValueError: return FALLBACK_PORT def _url_with_port(url: str, port: int) -> str: parts = urllib.parse.urlsplit(url) host = parts.hostname or "127.0.0.1" return urllib.parse.urlunsplit( (parts.scheme or "http", f"{host}:{port}", parts.path, "", "")) def update_config_api_url_port(port: int, config_path: Optional[Path] = None) -> bool: """Rewrite the port inside AUDIOCPP_API_URL in app/converter/config.py. Only the quoted URL literal is replaced; surrounding lines and the trailing comment are preserved. Returns True when the file was changed. """ path = Path(config_path) if config_path is not None else CONFIG_PATH try: text = path.read_text(encoding="utf-8") except OSError: return False match = re.search(r'(?m)^(\s*AUDIOCPP_API_URL\s*=\s*")([^"]*)(")', text) if not match: return False new_url = _url_with_port(match.group(2), port) if new_url == match.group(2): return False text = text[:match.start(2)] + new_url + text[match.end(2):] try: path.write_text(text, encoding="utf-8") except OSError: return False return True def update_server_config_port(port: int) -> bool: """Rewrite the 'port' in the audio.cpp checkout's server.json. Loads ``/server.json``, sets its ``port`` to PORT, and rewrites it with the same ``json.dump`` formatting the wizard uses. Returns True when the file now carries PORT (a no-op when it already does), and False when there is no checkout/server.json or the file cannot be read or written. """ checkout = find_local_checkout() if checkout is None: return False server_json = checkout / "server.json" if not server_json.exists(): return False try: data = json.loads(server_json.read_text(encoding="utf-8")) except (OSError, ValueError): return False if not isinstance(data, dict): return False if data.get("port") == port: return True data["port"] = port try: with server_json.open("w", encoding="utf-8") as handle: json.dump(data, handle, indent=2, ensure_ascii=False) handle.write("\n") except OSError: return False return True def update_config_model_ids(model_id: str, clone_model_id: Optional[str] = None, config_path: Optional[Path] = None) -> bool: """Rewrite AUDIOCPP_MODEL_ID (and AUDIOCPP_CLONE_MODEL_ID when given). Only the quoted id literals are replaced; surrounding lines and comments are preserved. Returns True when the file was changed. """ path = Path(config_path) if config_path is not None else CONFIG_PATH try: text = path.read_text(encoding="utf-8") except OSError: return False updates: List[Tuple[str, str]] = [("AUDIOCPP_MODEL_ID", model_id)] if clone_model_id is not None: updates.append(("AUDIOCPP_CLONE_MODEL_ID", clone_model_id)) changed = False for name, value in updates: match = re.search(r'(?m)^(\s*' + name + r'\s*=\s*")([^"]*)(")', text) if match and match.group(2) != value: text = text[:match.start(2)] + value + text[match.end(2):] changed = True if not changed: return False try: path.write_text(text, encoding="utf-8") except OSError: return False return True def default_model_id(family: str) -> str: """Derive a default server entry id from a family name.""" if family in PREFERRED_IDS: return PREFERRED_IDS[family] name = family if name.endswith("_tts"): name = name[:-4] return name.replace("_", "") or family def detect_audiocpp_dir() -> Optional[Path]: """Best-effort location of a local audio.cpp checkout with model_specs. Checks the AUDIOCPP_DIR environment variable, then ``app/audio.cpp`` in the tts-audiobook-generator root, then an ``audio.cpp`` directory in or above the current working directory. Returns the path only when it contains a ``model_specs`` directory. """ candidates: List[Path] = [] env_dir = os.environ.get("AUDIOCPP_DIR") if env_dir: candidates.append(Path(os.path.expanduser(env_dir))) candidates.append(APP_DIR / AUDIOCPP_DIR_NAME) cwd = Path.cwd() candidates.append(cwd / "audio.cpp") candidates.append(cwd.parent / "audio.cpp") candidates.append(cwd.parent.parent / "audio.cpp") for candidate in candidates: try: resolved = candidate.resolve() except OSError: continue if (resolved / "model_specs").is_dir(): return resolved return None # audio.cpp build directories are named ``--`` (e.g. # ``linux-cuda-release``, ``windows-vulkan-debug``, ``macos-metal-release``) # and the built server lands in ``/bin/audiocpp_server``. The Metal # macOS backend is reported as "cpu" here since it is not a separate # --backend choice for audiocpp_server. _BACKEND_TOKEN_RE = re.compile(r"-(cuda|vulkan|hip|cpu|metal)(?:-|$)") def detect_backend(audiocpp_dir: Path) -> Optional[str]: """Best-effort detection of the backend audiocpp_server was built for. Scans ``audiocpp_dir/build/*`` for build directories that contain a built ``bin/audiocpp_server`` (``.exe`` allowed on Windows) and reads the backend token out of the directory name (``-cuda-``, ``-vulkan-``, ``-hip-`` or ``-cpu-``; ``-metal-`` is mapped to ``cpu``). Returns the backend only when exactly one distinct backend was built, so a checkout with builds for several backends does not silently pick one. Returns None when there is no ``build/`` directory, no built server, or more than one distinct backend. """ build_root = audiocpp_dir / "build" if not build_root.is_dir(): return None backends: Set[str] = set() try: build_dirs = sorted(build_root.iterdir(), key=lambda p: p.name.lower()) except OSError: return None for build_dir in build_dirs: if not build_dir.is_dir(): continue server = build_dir / "bin" / "audiocpp_server" if not server.exists(): server_exe = build_dir / "bin" / "audiocpp_server.exe" if not server_exe.exists(): continue match = _BACKEND_TOKEN_RE.search(build_dir.name.lower()) if not match: continue token = match.group(1) backends.add("cpu" if token == "metal" else token) if len(backends) == 1: return next(iter(backends)) return None def _default_package(packages: List[dict]) -> Optional[dict]: """Pick the default package from a list of packages. Prefers the package flagged ``default: true``, then the first GGUF package, then the first package overall. Returns None for an empty list. """ if not packages: return None for package in packages: if package.get("default"): return package for package in packages: if package.get("format") == "gguf": return package return packages[0] def load_model_catalog(audiocpp_dir: Path) -> List[dict]: """Read model_specs/*.json and return the TTS-capable families. Each returned entry has: family, display_name, description, languages, clone_capable, packages (the full list from the spec), install_id (recommended package id), default_path (``models/``), and preferred_id. All families are treated equally and listed in alphabetical order by display name. """ specs_dir = audiocpp_dir / "model_specs" if not specs_dir.is_dir(): raise NotADirectoryError( f"{audiocpp_dir} has no model_specs/ directory; point " "--audiocpp-dir at an audio.cpp checkout") entries: List[dict] = [] for spec_path in sorted(specs_dir.glob("*.json")): try: spec = json.loads(spec_path.read_text(encoding="utf-8")) except (OSError, ValueError): continue tasks = spec.get("tasks") or [] if "tts" not in tasks and spec.get("category") != "tts": continue family = spec.get("family") or spec_path.stem packages = spec.get("packages") or [] package = _default_package(packages) if package is None: # No installable package: skip (cannot be hosted from a path). continue target_directory = package.get("target_directory") or family languages = spec.get("languages") or [] display_name = spec.get("display_name") or family description = spec.get("description") or "" entries.append({ "family": family, "display_name": display_name, "description": description, "languages": languages, "tasks": list(tasks), "clone_capable": "clone" in tasks, "packages": packages, "install_id": package.get("id") or family, "default_path": f"models/{target_directory}", "preferred_id": default_model_id(family), }) # All families are treated equally: alphabetical by display name. entries.sort(key=lambda entry: entry["display_name"].lower()) return entries def is_design_package(package: dict) -> bool: """Return True when a package's name marks it a voice-design model. audio.cpp voice-design packages (whose id, display name, or target directory mentions "voice design") are the only packages that must be hosted with task "vdes"; their role is not in the schema, only in those strings, so it is detected from them. """ text = " ".join(str(package.get(key, "")) for key in ("id", "display_name", "target_directory")) return bool(DESIGN_PACKAGE_RE.search(text)) def package_dir_options(entry: dict) -> List[dict]: """Return one option per distinct target_directory of a family's packages. Each option is a dict with: target_directory, install_id (the recommended package id inside that directory), design (voice-design package flag), and recommended (whether it holds the family's default package). Precisions that share a directory (q8_0/bf16/...) collapse to a single option. """ packages = entry.get("packages") or [] default_pkg = _default_package(packages) default_dir = (default_pkg or {}).get("target_directory") or entry["family"] by_dir: Dict[str, List[dict]] = {} order: List[str] = [] for package in packages: directory = package.get("target_directory") or entry["family"] if directory not in by_dir: by_dir[directory] = [] order.append(directory) by_dir[directory].append(package) options: List[dict] = [] for directory in order: package = _default_package(by_dir[directory]) options.append({ "target_directory": directory, "install_id": (package or {}).get("id") or directory, "design": is_design_package(package or {}), "recommended": directory == default_dir, }) # Put the recommended package first for a friendlier checklist. options.sort(key=lambda opt: not opt["recommended"]) return options def build_model_entry(family: str, model_id: str, model_path: str, task: str = TASK_TTS) -> dict: """Assemble one server.json model entry. ``task`` defaults to "tts"; voice design packages are hosted with "vdes" so the server runs its design session for speech requests (audiobook.py then requires --instructions with that entry). """ return { "id": model_id, "family": family, "path": model_path, "task": task, "mode": "offline", } def build_server_config(host: str, port: int, backend: str, lazy_load: bool, model_entries: List[dict], voice_dir: Optional[str] = None) -> dict: """Assemble the server.json document. ``voice_dir`` is a server-level cloning voice library; when set, every hosted clone-capable family can use its voices with ``--voice``. """ config_doc = { "host": host, "port": port, "backend": backend, "lazy_load": lazy_load, "models": model_entries, } if voice_dir: config_doc["voice_dir"] = voice_dir return config_doc def transcribe_wav_dir(wav_files: list, whisper_model: str, cancel=None) -> Dict[str, str]: """Transcribe each wav file and return a mapping of stem -> transcript. CANCEL (a ``threading.Event``) is checked between files so the in-TUI task view can stop a long transcription early. """ transcripts: Dict[str, str] = {} for wav_file in wav_files: if cancel is not None and cancel.is_set(): print("[INFO] Transcription cancelled") break name = wav_file.stem print(f"[INFO] Transcribing {wav_file.name} (voice '{name}')...") text = transcribe_reference_audio(str(wav_file), model_name=whisper_model) if text: print(f"[OK] {name}: {text}") else: print(f"[WARNING] No transcript for '{name}'; cloning works best " "with an accurate transcript — consider editing prompt_text " "by hand before starting the server") transcripts[name] = text or "" return transcripts def print_empty_transcript_warning(transcripts: Dict[str, str]) -> None: """Print a loud, final warning for voices whose transcript is empty.""" empty = sorted(name for name, text in transcripts.items() if not text) if not empty: return bar = "=" * 70 print() print(bar) print("[WARNING] MANUAL TRANSCRIPTION REQUIRED") print(bar) listing = " - " + "\n - ".join(empty) if len(empty) > 1 else f" - {empty[0]}" print(f"The following voice(s) have an EMPTY transcript in prompt_text:\n" f"{listing}") print("Those voices will NOT work until you add an accurate transcript.") print(f"Edit {PROMPT_TEXT_FILENAME} in your voice directory and fill in the " "text after '|' for each voice above.") print(bar) def _apply_port_sync(port: int, accepted: bool) -> None: """Write the port into app/converter/config.py, or report when declined.""" if accepted: if not update_config_api_url_port(port): print(f"[WARNING] Could not update {CONFIG_PATH}; edit " "AUDIOCPP_API_URL by hand so audiobook.py uses the " "new port") else: print("[WARNING] Left AUDIOCPP_API_URL unchanged; audiobook.py " f"will still use port {config_port()}") def _decide_transcription(wav_files: list, existing: Dict[str, str], prompt_exists: bool, force: bool, confirm: Callable[[str, bool], bool]) -> dict: """Decide which voices to transcribe; CONFIRM asks the plan questions. Returns a plan dict: {"mode": "all"|"missing"|"keep", "missing": [...], "existing": {...}} — "existing" carries the prompt_text mapping read while deciding, so the caller can reuse it instead of reading the file again. """ mode = "all" missing: List[Path] = [] if prompt_exists and not force: missing = [wav for wav in wav_files if not existing.get(wav.stem, "").strip()] if not missing: if confirm("All voices already transcribed in prompt_text. " "Re-transcribe anyway?", False): mode = "all" else: mode = "keep" elif confirm("Existing transcription and new .wavs detected, " "only transcribe new voices?", True): mode = "missing" else: mode = "all" return {"mode": mode, "missing": missing, "existing": existing} def _transcribe(args: argparse.Namespace, include_clone: bool, plan: dict, cancel=None) -> Tuple[Dict[str, str], bool]: """Transcribe the wav directory into a stem -> transcript mapping. Returns the mapping and a flag indicating whether it should be written to prompt_text (False when an existing, complete prompt_text is kept as-is). PLAN is always pre-collected — by the TUI (via _decide_transcription and its confirm callbacks) or by _flag_plan for a non-interactive run — so no questions are asked here. CANCEL is checked between files. """ if not include_clone: print(f"[WARNING] Ignoring {args.input_dir}: no clone-capable family " "selected, so voice presets are not used") return {}, False wav_files = find_wav_files(args.input_dir) if not wav_files: print(f"[WARNING] No .wav files found in {args.input_dir}; writing the " "config without a voice_dir") return {}, False prompt_path = args.input_dir / PROMPT_TEXT_FILENAME existing = plan.get("existing") or {} if plan else {} if plan["mode"] == "keep": print(f"[INFO] Kept existing {prompt_path}; all voices were " "already transcribed, nothing new to transcribe") return existing, False if whisper_backend_available() is None: print("[WARNING] Neither faster_whisper nor whisper was found, so " "reference .wav files cannot be transcribed automatically and " "every transcript will be empty.") print(" Install whisper (or faster_whisper) in your " "audiobook environment to transcribe automatically; otherwise " "transcripts must be added by hand (see the warning at the end).") if plan["mode"] == "missing": new_transcripts = transcribe_wav_dir(plan["missing"], args.whisper_model, cancel=cancel) transcripts = dict(existing) transcripts.update(new_transcripts) else: transcripts = transcribe_wav_dir(wav_files, args.whisper_model, cancel=cancel) return transcripts, True def _flag_plan(wav_files: list, prompt_path: Path, force: bool) -> dict: """Build a transcription plan for a non-interactive (flag-only) run. With --force everything is re-transcribed; otherwise an existing prompt_text is reused and only voices with an empty transcript are re-transcribed, mirroring what the TUI confirms interactively. """ if prompt_path.exists() and not force: existing = read_prompt_text(prompt_path) missing = [wav for wav in wav_files if not existing.get(wav.stem, "").strip()] if not missing: return {"mode": "keep", "missing": [], "existing": existing} return {"mode": "missing", "missing": missing, "existing": existing} return {"mode": "all", "missing": [], "existing": {}} def _offer_config_model_id_sync(model_id: str, accepted: Optional[bool]) -> None: """Point app/converter/config.py at a single hosted model entry. The converter requests the model id configured in AUDIOCPP_MODEL_ID, and single-model servers use the same id for the clone entry, so both ids are rewritten together. ACCEPTED is True/False (apply/skip the rewrite) or None when no single-entry sync applies (nothing to do). """ if config.AUDIOCPP_MODEL_ID == model_id \ and config.AUDIOCPP_CLONE_MODEL_ID == model_id: return if accepted is None: return if accepted: if not update_config_model_ids(model_id, model_id): print(f"[WARNING] Could not update {CONFIG_PATH}; edit " "AUDIOCPP_MODEL_ID and AUDIOCPP_CLONE_MODEL_ID by hand so " "audiobook.py uses this model") else: print("[WARNING] Left the model ids unchanged; audiobook.py will " f"still request model '{config.AUDIOCPP_MODEL_ID}'") def _build_entries(family_keys: List[str], chosen: Dict[str, List[dict]], catalog_by_family: Dict[str, dict], task_picker: Callable[[str], str], id_picker: Callable[[str, str, str], str], known_tasks: Optional[Dict[Tuple[str, str], str]] = None ) -> Tuple[List[dict], List[str], List[Tuple[str, str]], List[str], bool]: """Build server.json model entries from the selected families/packages. TASK_PICKER is called for each design package to choose vdes/tts; ID_PICKER resolves a duplicate server entry id. KNOWN_TASKS maps ``(family, target_directory)`` to a previously-stored task ("tts" or "vdes") so a modify run preserves how a design package was hosted instead of re-asking. Returns (model_entries, entry_ids, install_guidance, design_entry_ids, include_clone). """ model_entries: List[dict] = [] entry_ids: List[str] = [] install_guidance: List[Tuple[str, str]] = [] design_entry_ids: List[str] = [] include_clone = False for family in family_keys: entry = catalog_by_family[family] include_clone = include_clone or entry["clone_capable"] for opt in chosen[family]: if opt["design"]: task = known_tasks.get((family, opt["target_directory"])) \ if known_tasks else None if task is None: task = task_picker(opt["install_id"]) else: task = TASK_TTS base_id = (f"{entry['preferred_id']}-design" if task == TASK_VDES else entry["preferred_id"]) model_id = base_id if model_id in entry_ids: model_id = id_picker(entry["display_name"], opt["install_id"], f"{base_id}-2") entry_ids.append(model_id) model_entries.append(build_model_entry( family, model_id, f"models/{opt['target_directory']}", task=task)) install_guidance.append((entry["display_name"], opt["install_id"])) if task == TASK_VDES: design_entry_ids.append(model_id) return (model_entries, entry_ids, install_guidance, design_entry_ids, include_clone) def _write_and_advise(audiocpp_dir: Path, wav_dir: Optional[Path], output_path: Path, model_entries: List[dict], install_guidance: List[Tuple[str, str]], host: str, port: int, backend: str, lazy_load: bool, transcripts: Dict[str, str], write_prompt: bool) -> None: """Console phase shared by both UI modes: write files, print summary. After a successful run the console output is the path of the written server.json. The model install commands (and optional automatic download) are handled separately by _install_models, called by both UI modes once the user has decided whether to download. """ voice_dir: Optional[str] = None if transcripts: if write_prompt: prompt_path = wav_dir / PROMPT_TEXT_FILENAME write_prompt_text(wav_dir, transcripts) print(f"[OK] Wrote {prompt_path}") voice_dir = str(wav_dir.resolve()) server_config = build_server_config( host=host, port=port, backend=backend, lazy_load=lazy_load, model_entries=model_entries, voice_dir=voice_dir) with output_path.open("w", encoding="utf-8") as handle: json.dump(server_config, handle, indent=2, ensure_ascii=False) handle.write("\n") count = len(model_entries) print(f"Wrote {output_path.resolve()} with {count} " f"{'entry' if count == 1 else 'entries'}.") def _install_models(audiocpp_dir: Path, install_guidance: List[Tuple[str, str]], download: bool, emit=None, cancel=None) -> None: """Print and optionally run the model install commands. One ``python install `` command per hosted model (de-duped by install id). When DOWNLOAD is True each command is run in the audio.cpp checkout via ``subprocess`` so the models are downloaded automatically; a failing install is reported as a warning and does not abort the remaining downloads. When DOWNLOAD is False (or the model manager is missing) the commands are only printed, copy-pasteable as before. With EMIT given (the in-TUI task view) each download streams its output to EMIT and — when the checkout's ``model_manager_v2.py`` supports it — runs with ``--progress --cancel-file`` so the view can show a real byte progress bar and cancel gracefully. CANCEL aborts a running download. """ manager = audiocpp_dir / "tools" / "model_manager_v2.py" seen: Set[str] = set() install_ids: List[str] = [] for _, install_id in install_guidance: if install_id not in seen: seen.add(install_id) install_ids.append(install_id) supports_progress = emit is not None and _manager_supports_progress(manager) if download and not manager.is_file(): print(f"[WARNING] {manager} not found; printing the install commands " "instead of running them") download = False for install_id in install_ids: command = f"python {manager} install {install_id}" if not download: print(command) continue print(f"[INFO] Downloading {install_id}...") argv = [sys.executable, str(manager), "install", install_id] cancel_file: Optional[Path] = None on_cancel = None if supports_progress: fd, cancel_path = tempfile.mkstemp( prefix="audiocpp_cancel_", suffix=".cancel") os.close(fd) cancel_file = Path(cancel_path) cancel_file.unlink() # absent = not cancelled argv += ["--progress", "--cancel-file", str(cancel_file)] on_cancel = cancel_file.touch try: rc = common.run_console_subprocess( argv, cwd=str(audiocpp_dir), emit=emit, cancel=cancel, on_cancel=on_cancel) except OSError as exc: print(f"[WARNING] Could not run {command}: {exc}") rc = 1 finally: if cancel_file is not None: try: cancel_file.unlink() except OSError: pass if rc != 0: print(f"[WARNING] install {install_id} exited with code " f"{rc}; the model may need to be downloaded " "by hand") def _manager_supports_progress(manager: Path) -> bool: """True when MANAGER (model_manager_v2.py) supports --progress output. The ``--progress``/``--cancel-file`` flags are relatively recent; an older audio.cpp checkout may not have them, so probe the script source once instead of failing the download with an unknown flag. """ try: text = manager.read_text(encoding="utf-8", errors="ignore") except OSError: return False return "AUDIOCPP_PROGRESS" in text and "--cancel-file" in text def _decide_download(audiocpp_dir: Path, confirm: Callable[[str, bool], bool]) -> bool: """Ask whether to download the selected models now. CONFIRM asks the yes/no question (ask_bool for the line prompts, a TUI confirm for the wizard). When the audio.cpp model manager is missing the prompt is skipped and False is returned, so the install commands are only printed rather than offered to run. """ manager = audiocpp_dir / "tools" / "model_manager_v2.py" if not manager.is_file(): return False return confirm( "Automatically download the selected models with model_manager_v2.py " "now?", False) def _build_tree_families(catalog: List[dict]) -> List[dict]: """Shape the catalog into the checkbox_tree widget's family list.""" families: List[dict] = [] for entry in catalog: capabilities = ["tts"] if "clone" in entry["tasks"]: capabilities.append("cloning") if "design" in entry["tasks"]: capabilities.append("design") name = entry["display_name"] options = [] for opt in package_dir_options(entry): options.append({ "key": opt["target_directory"], "label": opt["install_id"], "recommended": opt["recommended"], }) families.append({ "label": name, "detail": ", ".join(capabilities), "options": options, }) return families def _wizard(stdscr, args: argparse.Namespace, parser: argparse.ArgumentParser ) -> Optional[dict]: """Run every TUI screen; return the collected settings, or None to abort. The wizard is driven by ``tui.Wizard`` as a stack of screen closures: each screen shows one interactive widget and returns the next screen (a closure), ``Wizard.BACK`` (Esc/q pressed — pop to the previous screen), or the final settings dict. Only screens that actually render are pushed, so Esc always lands on the previous real screen. A step whose value is already provided by a flag (``--host``, ``--port``, ``--families``, ...) or does not apply (e.g. the port-sync prompt when the port did not change) is folded into the ``_after_*`` guards and never becomes a screen. Esc on the first screen aborts the whole wizard. """ s: dict = {} def ask_confirm(question: str, default: bool) -> bool: result = tui.confirm(stdscr, question, default=default, cancel_value=_GO_BACK) if result is _GO_BACK: raise _GoBack() return result def resolve_checkout(audiocpp_dir: Path) -> None: """Validate AUDIOCPP_DIR and populate the wizard state ``s``.""" audiocpp_dir = Path(audiocpp_dir).resolve() if not audiocpp_dir.is_dir(): raise _TuiError(f"audio.cpp checkout not found: " f"{audiocpp_dir}") root = _resolve_audiocpp_root(audiocpp_dir) if root is None: raise _TuiError( f"{audiocpp_dir} has no model_specs/ directory; " "select the root of your audio.cpp checkout") audiocpp_dir = root try: catalog = load_model_catalog(audiocpp_dir) except NotADirectoryError as exc: raise _TuiError(str(exc)) if not catalog: raise _TuiError(f"No TTS model families found in " f"{audiocpp_dir}/model_specs; check the " "checkout is up to date") catalog_by_family = {entry["family"]: entry for entry in catalog} output_path = args.output if args.output is not None \ else audiocpp_dir / "server.json" # Modify flow: an existing server.json seeds the wizard's screens # instead of being overwritten from scratch (an explicit --force # still starts fresh). existing_config = load_server_config(output_path) \ if not args.force else None if existing_config is not None: existing_selected, existing_tasks = \ server_config_selections(existing_config, catalog) else: existing_selected, existing_tasks = {}, {} s.update({ "audiocpp_dir": audiocpp_dir, "catalog": catalog, "catalog_by_family": catalog_by_family, "output_path": output_path, "existing_config": existing_config, "existing_selected": existing_selected, "existing_tasks": existing_tasks, "existing_host": existing_config.get("host") if existing_config else None, "existing_port": existing_config.get("port") if existing_config else None, "existing_backend": existing_config.get("backend") if existing_config else None, "existing_voice_dir": existing_config.get("voice_dir") if existing_config else None, "detected_backend": detect_backend(audiocpp_dir), }) def _families_from_flag() -> None: requested = [f.strip() for f in args.families.split(",") if f.strip()] unknown = [f for f in requested if f not in s["catalog_by_family"]] if unknown: raise _TuiError( f"Unknown family in --families: {', '.join(unknown)}. " f"Available: {', '.join(s['catalog_by_family'])}") chosen: Dict[str, List[dict]] = {} family_keys: List[str] = [] for family in requested: if family not in family_keys: family_keys.append(family) chosen[family] = [opt for opt in package_dir_options( s["catalog_by_family"][family]) if opt["recommended"]] s["chosen"] = chosen s["family_keys"] = family_keys def _compute_entries() -> None: # Design task menus and duplicate-id renames. Esc on any of them # raises _GoBack, which the caller turns into Wizard.BACK (the # design/duplicate-id prompts are grouped: Esc returns to the # families tree). def task_picker(install_id: str) -> str: result = tui.menu( stdscr, f"How should the '{install_id}' package be hosted?", [ ("design (vdes) - describe the voice with " "--instructions", TASK_VDES), ("tts - normal synthesis", TASK_TTS), ], default_index=0, back_value=_GO_BACK) if result is _GO_BACK: raise _GoBack() return result def id_picker(display_name: str, install_id: str, default: str) -> str: result = tui.line_edit( stdscr, f"Server model id for {display_name} package " f"'{install_id}'", default, back_value=_GO_BACK) if result is _GO_BACK: raise _GoBack() return result model_entries, entry_ids, install_guidance, \ design_entry_ids, include_clone = _build_entries( s["family_keys"], s["chosen"], s["catalog_by_family"], task_picker, id_picker, known_tasks=s["existing_tasks"]) s.update({ "model_entries": model_entries, "entry_ids": entry_ids, "install_guidance": install_guidance, "design_entry_ids": design_entry_ids, "include_clone": include_clone, }) def _finalize() -> dict: return { "audiocpp_dir": s["audiocpp_dir"], "catalog": s["catalog"], "catalog_by_family": s["catalog_by_family"], "output_path": s["output_path"], "family_keys": s["family_keys"], "chosen": s["chosen"], "model_entries": s["model_entries"], "entry_ids": s["entry_ids"], "install_guidance": s["install_guidance"], "design_entry_ids": s["design_entry_ids"], "include_clone": s["include_clone"], "host": s["host"], "port": s["port"], "backend": s["backend"], "build": s["build"], "lazy_load": s["lazy_load"], "sync_port": s["sync_port"], "sync_model_ids": s["sync_model_ids"], "wav_dir": s["wav_dir"], "plan": s["plan"], "download": s["download"], "delete_unused": s["delete_unused"], "unused_entries": s["unused_entries"], } def screen_families(): """Pick TTS model families and packages (the modify tree).""" tree_families = _build_tree_families(s["catalog"]) # Modify flow: pre-check the models an existing server.json hosts, # so the tree opens as a "modify" list rather than a fresh one. checked_set = set() for family, dirs in s["existing_selected"].items(): if family not in s["catalog_by_family"]: continue family_index = s["catalog"].index(s["catalog_by_family"][family]) valid_dirs = {opt["target_directory"] for opt in package_dir_options( s["catalog_by_family"][family])} for target in dirs: if target in valid_dirs: checked_set.add((family_index, target)) picked = tui.checkbox_tree( stdscr, "Select TTS model families to host", tree_families, expand_all=args.all_packages, back_value=_GO_BACK, checked=checked_set) if picked is _GO_BACK: return tui.Wizard.BACK chosen: Dict[str, List[dict]] = {} family_keys: List[str] = [] for family_index, option_key in picked: family = s["catalog"][family_index]["family"] if family not in chosen: chosen[family] = [] family_keys.append(family) chosen[family].append(option_key) for family in list(chosen): keyed = {opt["target_directory"]: opt for opt in package_dir_options( s["catalog_by_family"][family])} chosen[family] = [keyed[key] for key in chosen[family]] s["chosen"] = chosen s["family_keys"] = family_keys return screen_host def _after_families(): if args.families is not None: _families_from_flag() return screen_host return screen_families def screen_host(): """Build the model entries, then ask the bind host. The task/id pickers (when any) run here too and are grouped with this screen: Esc on one of them (or on the host field) returns to the families tree. """ try: _compute_entries() except _GoBack: return tui.Wizard.BACK if args.host is not None: s["host"] = args.host return _after_host() host = tui.line_edit( stdscr, "Bind host", s["existing_host"] if isinstance(s["existing_host"], str) else DEFAULT_HOST, help_lines=["The IP address audiocpp will be hosted on", "127.0.0.1 (this machine) is probably " "correct"], back_value=_GO_BACK) if host is _GO_BACK: return tui.Wizard.BACK s["host"] = host return _after_host() def _after_host(): if args.port is None: return screen_port s["port"] = args.port return _after_port() def screen_port(): port_text = tui.line_edit( stdscr, "Port", str(s["existing_port"]) if isinstance(s["existing_port"], int) else str(config_port()), validate=lambda s: None if (s.isdigit() and 1 <= int(s) <= 65535) else "Enter a port number between 1 and 65535", help_lines=["The port audiocpp will be hosted on"], back_value=_GO_BACK) if port_text is _GO_BACK: return tui.Wizard.BACK s["port"] = int(port_text) return _after_port() def _after_port(): s["sync_port"] = None if s["port"] != config_port(): return screen_sync_port return _after_sync() def screen_sync_port(): sync_port = tui.confirm( stdscr, "Update AUDIOCPP_API_URL in app/converter/config.py " f"to port {s['port']} so audiobook.py talks to this server", default=True, cancel_value=_GO_BACK) if sync_port is _GO_BACK: return tui.Wizard.BACK s["sync_port"] = sync_port return _after_sync() def _after_sync(): if args.build_backend: s["backend"] = args.build_backend s["build"] = s["detected_backend"] is None return _after_backend() if args.backend: s["backend"] = args.backend s["build"] = False return _after_backend() if s["detected_backend"] is not None: # Already built: use the detected backend, no menu, no build. s["backend"] = s["detected_backend"] s["build"] = False return _after_backend() # Not built for any backend yet: always ask which backend the server # should use and offer to build it — even on a modify run, so a user # who declined the build the first time is never stranded without a # way to build from the TUI. return screen_backend def screen_backend(): # Pre-select the backend an existing server.json records (modify # flow), so re-running setup lands on the previous choice. backend_options, backend_default = _backend_options(None) if s["existing_backend"] in BACKENDS: backend_default = next( (index for index, (_label, value) in enumerate(backend_options) if value == s["existing_backend"]), backend_default) backend = tui.menu( stdscr, "Which inference backend should audiocpp_server " "use?", backend_options, default_index=backend_default, back_value=_GO_BACK) if backend is _GO_BACK: return tui.Wizard.BACK s["backend"] = backend if built_server_binary(s["audiocpp_dir"], backend) is not None: # A checkout with builds for several backends: this one is # already built, so there is nothing to build. s["build"] = False return _after_backend() return screen_build def screen_build(): # Not built for the chosen backend yet: offer to build it now. The # build itself runs in the TUI task view (or the console tail for # CLI runs) after the wizard. build = tui.confirm( stdscr, f"audiocpp_server is not built for {s['backend']}. " f"Build it now (runs scripts/build_*)?", default=True, cancel_value=_GO_BACK) if build is _GO_BACK: return tui.Wizard.BACK s["build"] = build return _after_backend() def _after_backend(): s["lazy_load"] = True return _after_lazy() def _after_lazy(): if args.input_dir is not None: s["wav_dir"] = args.input_dir return _after_wav() if s["include_clone"]: return screen_wav s["wav_dir"] = None return _after_wav() def screen_wav(): wav_start = detect_wav_dir(s["audiocpp_dir"], TTS_ROOT) # Modify flow: an existing voice_dir seeds the browser so the user # can accept it on Enter instead of re-navigating. if isinstance(s["existing_voice_dir"], str) and s["existing_voice_dir"]: wav_start = Path(s["existing_voice_dir"]) wav_dir = tui.browse_directory( stdscr, "Select the directory with your .wav voices", info=_wav_dir_info, preview=_wav_dir_preview, start=wav_start if wav_start is not None else VOICES_DIR, back_value=_GO_BACK) if wav_dir is _GO_BACK: return tui.Wizard.BACK s["wav_dir"] = wav_dir return _after_wav() def _after_wav(): s["plan"] = None if s["include_clone"] and s["wav_dir"] is not None: wav_files = find_wav_files(s["wav_dir"]) if wav_files: prompt_path = s["wav_dir"] / PROMPT_TEXT_FILENAME if prompt_path.exists() and not args.force: return screen_transcription existing = read_prompt_text(prompt_path) if ( prompt_path.exists() and not args.force) else {} s["plan"] = _decide_transcription( wav_files, existing, prompt_path.exists(), args.force, ask_confirm) return _after_transcription() def screen_transcription(): # Transcription plan (questions only; transcription runs after). wav_files = find_wav_files(s["wav_dir"]) prompt_path = s["wav_dir"] / PROMPT_TEXT_FILENAME existing = read_prompt_text(prompt_path) if ( prompt_path.exists() and not args.force) else {} try: s["plan"] = _decide_transcription( wav_files, existing, prompt_path.exists(), args.force, ask_confirm) except _GoBack: return tui.Wizard.BACK return _after_transcription() def _after_transcription(): s["sync_model_ids"] = None if len(s["entry_ids"]) == 1 and not ( config.AUDIOCPP_MODEL_ID == s["entry_ids"][0] and config.AUDIOCPP_CLONE_MODEL_ID == s["entry_ids"][0]): return screen_model_sync return _after_model_sync() def screen_model_sync(): sync_model_ids = tui.confirm( stdscr, "Update AUDIOCPP_MODEL_ID and " "AUDIOCPP_CLONE_MODEL_ID in app/converter/config.py to " f"'{s['entry_ids'][0]}' so audiobook.py uses this model", default=True, cancel_value=_GO_BACK) if sync_model_ids is _GO_BACK: return tui.Wizard.BACK s["sync_model_ids"] = sync_model_ids return _after_model_sync() def _after_model_sync(): new_paths = {entry["path"] for entry in s["model_entries"]} s["unused_entries"] = unused_installed_entries( s["output_path"], new_paths) \ if s["existing_config"] is not None else [] s["delete_unused"] = False if s["unused_entries"]: return screen_delete_unused return _after_delete() def screen_delete_unused(): delete_unused = tui.confirm( stdscr, "Delete unused models?", default=False, cancel_value=_GO_BACK) if delete_unused is _GO_BACK: return tui.Wizard.BACK s["delete_unused"] = delete_unused return _after_delete() def _after_delete(): manager = s["audiocpp_dir"] / "tools" / "model_manager_v2.py" if manager.is_file(): return screen_download s["download"] = False return _finalize() def screen_download(): # Automatic model download (or print the install commands). try: s["download"] = _decide_download(s["audiocpp_dir"], ask_confirm) except _GoBack: return tui.Wizard.BACK return _finalize() # First screen: resolve the checkout directly when it already exists # (the modify flow), so the wizard starts on a real screen. When no # checkout exists, clone it into ./app/audio.cpp (streaming inside the # TUI task view, not by dropping to the console) without asking, then # continue the same way. audiocpp_dir = args.audiocpp_dir if audiocpp_dir is None: audiocpp_dir = find_local_checkout() if audiocpp_dir is None: target = APP_DIR / AUDIOCPP_DIR_NAME rc = taskview.run_steps(stdscr, "Clone audio.cpp", [taskview.TaskStep( f"Cloning audio.cpp into {target}", lambda emit, cancel: common.git_clone( AUDIOCPP_GIT_URL, target, emit=emit, cancel=cancel))]) if rc == 130: # Cancelled from the task view: abort the wizard quietly. return None if rc != 0: raise _TuiError( f"git clone failed (exit {rc}). Clone " f"audio.cpp manually: git clone " f"{AUDIOCPP_GIT_URL} {target}") audiocpp_dir = target resolve_checkout(audiocpp_dir) first = _after_families() return tui.Wizard().run(first) def load_server_config(server_json: Path) -> Optional[dict]: """Read server.json into a dict, or None when it cannot be used. Returns None for a missing file, unreadable content, or a non-dict document. Used by the wizard's modify flow to pre-fill its screens from an existing config instead of prompting to overwrite it. """ if not server_json.exists(): return None try: data = json.loads(server_json.read_text(encoding="utf-8")) except (OSError, ValueError): return None if not isinstance(data, dict): return None return data def server_config_selections(server_config: dict, catalog: List[dict] ) -> Tuple[Dict[str, List[str]], Dict[Tuple[str, str], str]]: """Map an existing server.json's models back to catalog selections. Returns ``(selected_dirs, tasks)``: ``selected_dirs`` maps a catalog family to the target directories it hosts (``models/`` paths with the ``models/`` prefix stripped, in server.json order), and ``tasks`` maps ``(family, target_directory)`` to the entry's task (``"tts"`` or ``"vdes"``) so the wizard can preserve how design packages were hosted. Entries whose family is not in the CATALOG are ignored — the wizard cannot offer them again. """ families = {entry["family"] for entry in catalog} selected_dirs: Dict[str, List[str]] = {} tasks: Dict[Tuple[str, str], str] = {} for entry in server_config.get("models") or []: if not isinstance(entry, dict): continue family = entry.get("family") if not isinstance(family, str) or family not in families: continue path = entry.get("path") if not isinstance(path, str): continue target = path[len("models/"):] if path.startswith("models/") else path if family not in selected_dirs: selected_dirs[family] = [] if target not in selected_dirs[family]: selected_dirs[family].append(target) tasks[(family, target)] = str(entry.get("task") or TASK_TTS) return selected_dirs, tasks def _model_path_present(path: Path) -> bool: """True when a server.json model path holds actual model files. A present path is either a file (a single-model package) or a non-empty directory (the usual GGUF package target directory; an empty one means a download that never ran or was cleaned up halfway). """ try: if path.is_file(): return True if path.is_dir(): return any(path.iterdir()) except OSError: return False return False def missing_model_entries(server_json: Path) -> List[dict]: """Return the server.json model entries whose files are not on disk. Paths resolve exactly like audiocpp_server resolves them (relative paths against the server.json's directory). Each returned entry carries the entry ``id`` and ``rel`` (the configured path string); used by ``detect`` to warn that a conversion would fail until the models are installed. """ try: data = json.loads(server_json.read_text(encoding="utf-8")) except (OSError, ValueError): return [] if not isinstance(data, dict): return [] base = server_json.parent missing: List[dict] = [] for entry in data.get("models") or []: if not isinstance(entry, dict): continue rel = entry.get("path") if not isinstance(rel, str) or not rel: continue path = Path(rel) if Path(rel).is_absolute() else base / rel if _model_path_present(path): continue missing.append({"id": str(entry.get("id") or rel), "rel": rel}) return missing def _install_id_by_path(audiocpp_dir: Path) -> Dict[str, str]: """Map ``models/`` -> catalog install id. The catalog package that installs a model is derived from the ``default_path`` of each TTS family; an entry whose path matches no catalog package has no install id. """ by_path: Dict[str, str] = {} try: for entry in load_model_catalog(audiocpp_dir): by_path[entry["default_path"]] = entry["install_id"] except (NotADirectoryError, OSError): pass return by_path def installed_model_entries(server_json: Path) -> List[dict]: """Return the server.json model entries whose files ARE on disk. The complement of ``missing_model_entries``: each returned entry carries the entry ``id`` and ``rel`` (the configured path string), resolved exactly like ``missing_model_entries`` (relative against the server.json's directory). Used by the wizard's "Delete unused models?" step to find already-downloaded models that were unselected. """ try: data = json.loads(server_json.read_text(encoding="utf-8")) except (OSError, ValueError): return [] if not isinstance(data, dict): return [] base = server_json.parent installed: List[dict] = [] for entry in data.get("models") or []: if not isinstance(entry, dict): continue rel = entry.get("path") if not isinstance(rel, str) or not rel: continue path = Path(rel) if Path(rel).is_absolute() else base / rel if _model_path_present(path): installed.append({"id": str(entry.get("id") or rel), "rel": rel}) return installed def missing_model_install_guidance(audiocpp_dir: Path, missing: List[dict]) -> List[Tuple[str, str]]: """Map MISSING model entries to (display name, install id) pairs. The install id is derived from each entry's configured path via the catalog (see ``_install_id_by_path``); entries whose path matches no catalog package are skipped (there is no ``model_manager_v2.py install`` command for them). Feeds ``_install_models`` for the "Download Missing Models" action. """ by_path = _install_id_by_path(audiocpp_dir) guidance: List[Tuple[str, str]] = [] for item in missing: install_id = by_path.get(item["rel"]) if install_id: guidance.append((item["id"], install_id)) return guidance def model_install_hints(audiocpp_dir: Path, missing: List[dict]) -> List[str]: """Remediation lines for MISSING model entries (see missing_model_entries). Maps each entry's configured path back to the catalog package that installs it (``models/`` -> install id) so the line carries the exact ``model_manager_v2.py install`` command; entries whose directory matches no catalog package just name the path. """ by_path = _install_id_by_path(audiocpp_dir) hints: List[str] = [] for item in missing: install_id = by_path.get(item["rel"]) hint = f"model not downloaded: {item['id']} ({item['rel']})" if install_id: hint += (f" — install with: python tools/model_manager_v2.py " f"install {install_id}") hints.append(hint) return hints def install_models(audiocpp_dir: Path, guidance: List[Tuple[str, str]], emit=None, cancel=None) -> None: """Download the (display name, install id) models via the helper script. Runs ``model_manager_v2.py install`` for each de-duped install id in the checkout, streaming to the console (or to EMIT, the in-TUI task view); a failing install is reported as a warning and does not abort the rest. Used by the hub's "Download Missing Models" action (see ``missing_model_install_guidance`` for the mapping). """ _install_models(audiocpp_dir, guidance, download=True, emit=emit, cancel=cancel) def hand_install_guidance(audiocpp_dir: Path, missing: List[dict]) -> str: """Explain how to install MISSING model entries by hand. Returns a multi-line message listing each missing model's id and the path its files must be placed in (``rel``, resolved against the AUDIOCPP_DIR checkout). Used when the missing models cannot be mapped to a ``model_manager_v2.py install`` command, so the user still knows what to download and where to put it. """ lines = [ "None of the missing models map to a model_manager_v2.py install " "command.", "Download them by hand and place the files at these paths:", ] for item in missing: lines.append(f" {item['id']} -> {item['rel']}") lines.append(f"(paths are relative to {audiocpp_dir})") return "\n".join(lines) def unused_installed_entries(server_json: Path, new_paths: Set[str]) -> List[dict]: """Return installed server.json entries whose path is not in NEW_PATHS. The already-downloaded models (see ``installed_model_entries``) that the new selection does not host any more — the candidates for the wizard's "Delete unused models?" prompt. Entries whose files are not on disk are never listed (there is nothing to delete). """ return [entry for entry in installed_model_entries(server_json) if entry["rel"] not in new_paths] def delete_model_files(server_json: Path, entries: List[dict]) -> int: """Remove the on-disk model files for ENTRIES ({id, rel}) from disk. Each entry's ``rel`` is resolved exactly like the server resolves it (relative against ``server_json``'s directory; absolute paths honored), then removed as a directory tree or a single file. Missing entries are ignored. Returns the number of paths removed. Used by the wizard's "Delete unused models?" step — the regenerated server.json already only lists the kept models, so no entry cleanup is needed here. """ base = server_json.parent removed = 0 for item in entries: rel = item.get("rel") if not isinstance(rel, str) or not rel: continue path = Path(rel) if Path(rel).is_absolute() else base / rel try: if not path.exists(): continue if path.is_dir(): shutil.rmtree(path, ignore_errors=True) else: path.unlink() except OSError as exc: print(f"[WARNING] Could not remove {path}: {exc}") continue print(f"[OK] Removed unused model {path}") removed += 1 return removed def uninstall() -> int: """Remove the audio.cpp backend entirely: stop its server, delete the checkout. The checkout (``app/audio.cpp``, or wherever ``find_local_checkout`` resolves it) holds the built binary, the downloaded models, and the server.json, so removing the directory uninstalls the backend. A running server this tool started is stopped first (best-effort). Returns the exit code. """ servers.stop("audiocpp") checkout = find_local_checkout() if checkout is None: print("[INFO] No audio.cpp checkout to remove.") return 0 print(f"[INFO] Removing audio.cpp checkout {checkout}...") shutil.rmtree(checkout, ignore_errors=True) print("[OK] audio.cpp removed.") return 0 def find_local_checkout() -> Optional[Path]: """Best-effort location of an audio.cpp checkout with model_specs. Checks the AUDIOCPP_DIR environment variable, then ``app/audio.cpp`` inside the tts-audiobook-generator root, then an ``audio.cpp`` directory in or above the current working directory. Returns the path only when it contains a ``model_specs`` directory. """ candidates: List[Path] = [] env_dir = os.environ.get("AUDIOCPP_DIR") if env_dir: candidates.append(Path(os.path.expanduser(env_dir))) candidates.append(APP_DIR / AUDIOCPP_DIR_NAME) cwd = Path.cwd() candidates.append(cwd / AUDIOCPP_DIR_NAME) candidates.append(cwd.parent / AUDIOCPP_DIR_NAME) candidates.append(cwd.parent.parent / AUDIOCPP_DIR_NAME) for candidate in candidates: try: resolved = candidate.resolve() except OSError: continue if (resolved / "model_specs").is_dir(): return resolved return None def fetch_server_models(api_url: str) -> Optional[List[Dict[str, str]]]: """List a running audiocpp_server's model entries via GET /v1/models. Returns ``[{id, family, task}, ...]`` — the same shape the converter's client resolves at startup — or None when URL does not answer with a valid document (wrong server, still starting, older audio.cpp). Used by the hub to drive the convert menus against a remote server that has no local server.json describing it. """ try: with urllib.request.urlopen( f"{api_url.rstrip('/')}/v1/models", timeout=10) as response: payload = json.loads(response.read().decode("utf-8")) except (OSError, ValueError): # URLError/HTTPError/socket errors are OSErrors; a non-JSON body is # a ValueError. Anything else means "not an audiocpp_server". return None entries = payload.get("data") if isinstance(payload, dict) else None models: List[Dict[str, str]] = [] for entry in entries or []: if isinstance(entry, dict) and entry.get("id"): models.append({ "id": str(entry["id"]), "family": str(entry.get("family") or ""), "task": str(entry.get("task") or ""), }) return models def fetch_server_voices(api_url: str, model_id: str) -> Optional[List[str]]: """List a running audiocpp_server's voices for MODEL_ID. Queries ``GET /v1/audio/voices?model=`` — the endpoint the converter validates ``--voice`` against — and returns its voice-name list, or None when the server cannot be queried. Lets the hub offer a remote server's voices without reading its configuration locally. """ query = urllib.parse.urlencode({"model": model_id}) try: with urllib.request.urlopen( f"{api_url.rstrip('/')}/v1/audio/voices?{query}", timeout=10) as response: payload = json.loads(response.read().decode("utf-8")) except (OSError, ValueError): return None voices = payload.get("voices") if isinstance(payload, dict) else None if not isinstance(voices, list): return None return [str(voice) for voice in voices] def find_audiocpp_server_bin(audiocpp_dir: Path) -> Optional[Path]: """Return the built audiocpp_server binary, or None when not built. Scans ``audiocpp_dir/build/*`` for a build directory containing ``bin/audiocpp_server`` (``.exe`` allowed on Windows). When several builds exist the first (alphabetical) is returned. """ build_root = audiocpp_dir / "build" if not build_root.is_dir(): return None try: build_dirs = sorted(build_root.iterdir(), key=lambda p: p.name.lower()) except OSError: return None for build_dir in build_dirs: if not build_dir.is_dir(): continue for name in ("audiocpp_server", "audiocpp_server.exe"): server = build_dir / "bin" / name if server.exists(): return server return None def built_server_binary(audiocpp_dir: Path, backend: str) -> Optional[Path]: """Return the built audiocpp_server for BACKEND, or None. Like ``find_audiocpp_server_bin`` but limited to build directories whose name carries the BACKEND token (``-cuda-``, ``-vulkan-``, ``-hip-``, ``-cpu-``; ``-metal-`` counts as ``cpu``). A checkout with builds for several backends is asked which one to use without re-offering a build for a backend that is already built. """ build_root = audiocpp_dir / "build" if not build_root.is_dir(): return None try: build_dirs = sorted(build_root.iterdir(), key=lambda p: p.name.lower()) except OSError: return None for build_dir in build_dirs: if not build_dir.is_dir(): continue match = _BACKEND_TOKEN_RE.search(build_dir.name.lower()) if not match: continue token = "cpu" if match.group(1) == "metal" else match.group(1) if token != backend: continue for name in ("audiocpp_server", "audiocpp_server.exe"): server = build_dir / "bin" / name if server.exists(): return server return None def find_build_script(audiocpp_dir: Path) -> Optional[Path]: """Return the audio.cpp build helper script to run, or None. Prefers ``scripts/build_linux.sh``; otherwise the first ``scripts/build_*.sh`` it finds. (Windows ``.bat`` scripts are not run automatically — build manually there.) """ scripts = audiocpp_dir / "scripts" if not scripts.is_dir(): return None preferred = scripts / "build_linux.sh" if preferred.exists(): return preferred try: candidates = sorted(scripts.glob("build_*.sh"), key=lambda p: p.name.lower()) except OSError: return None return candidates[0] if candidates else None def build_audiocpp(audiocpp_dir: Path, backend: str, *, emit=None, cancel=None) -> int: """Build audiocpp_server for BACKEND, streaming output. With EMIT None the build script runs on the console (inherits the terminal); with EMIT given (the in-TUI task view) its output streams line by line to EMIT so the view can show progress, and CANCEL aborts it. On the EMIT (TUI) path the build output is also tee'd to ``app/logs/audiocpp_build_.log`` so it survives the curses session; when the build fails (and was not cancelled) a post-TUI notice with the copy-pastable command and the log path is queued for the console (see ``backends.common.record_post_tui_notice``). Returns the build script's exit code (non-zero when the script is missing). """ script = find_build_script(audiocpp_dir) if script is None: message = (f"[ERROR] No build script found in {audiocpp_dir}/scripts; " "build audiocpp_server manually (see the audio.cpp README)") print(message) if emit is not None: common.record_post_tui_notice(message) return 1 argv = ["sh", str(script), "--backend", backend, "--target", "audiocpp_server"] command = f"cd {audiocpp_dir} && {shlex.join(argv)}" if emit is None: print(f"[INFO] Building audiocpp_server for {backend} ({command})...") return common.run_console_subprocess(argv, cwd=audiocpp_dir) return _build_audiocpp_tui(emit, cancel, argv, command, audiocpp_dir) def _build_audiocpp_tui(emit, cancel, argv: List[str], command: str, audiocpp_dir: Path) -> int: """Run the build on the TUI path: tee output to a log file. Every emitted line is also written (and flushed) to ``app/logs/audiocpp_build_.log``. On failure a summary (the copy-pastable COMMAND and the log path) is emitted into the TUI, written to the log, and queued as a post-TUI console notice. A cancelled build (CANCEL set) is not reported as a failure, but its partial output stays in the log file. """ log_path = common.LOG_DIR / ( f"audiocpp_build_{datetime.now():%Y%m%d_%H%M%S}.log") log_path.parent.mkdir(parents=True, exist_ok=True) log_handle = log_path.open("w", encoding="utf-8") def tee(line: str) -> None: log_handle.write(line + "\n") log_handle.flush() emit(line) tee(f"[INFO] Building audiocpp_server ({command})...") rc = 0 try: rc = common.run_console_subprocess( argv, cwd=audiocpp_dir, emit=tee, cancel=cancel) if rc != 0 and (cancel is None or not cancel.is_set()): notice = (f"[ERROR] audio.cpp build failed (exit code {rc}).\n" f" Build log: {log_path}\n" f" Troubleshoot by re-running this command:\n" f" {command}") for line in notice.splitlines(): tee(line) common.record_post_tui_notice(notice) finally: log_handle.close() return rc def _print_launch_hint(audiocpp_dir: Path, output_path: Path) -> None: """Print the exact command to start the server (or build guidance). The command is prefixed with ``cd &&`` because the server discovers model_specs/.json relative to its working directory. """ binary = find_audiocpp_server_bin(audiocpp_dir) print() if binary is not None: print("Start the server with:") print(f" cd {audiocpp_dir} && {binary} --config {output_path}") else: print("[INFO] audiocpp_server binary not found. Build it first, e.g.:") script = find_build_script(audiocpp_dir) if script is not None: print(f" sh {script} --backend " "--target audiocpp_server") print(f" then run: cd {audiocpp_dir} && ./build/-" f"-release/bin/audiocpp_server --config {output_path}") def _execute_steps(settings: dict, args: argparse.Namespace) -> List[taskview.TaskStep]: """Build the ordered setup steps for the in-TUI task view. The same work ``_execute`` runs on the console, split into named steps so the view can show per-step state (build / transcribe / write / download) and progress. Shared results (the transcription mapping) travel through a small closure dict. Each step's ``work(emit, cancel)`` returns its exit code; subprocess steps stream through EMIT and abort on CANCEL, while print()-based steps are captured by the view's stdout redirect. """ audiocpp_dir = settings["audiocpp_dir"] state: dict = {} steps: List[taskview.TaskStep] = [] if settings.get("build"): def build(emit, cancel): rc = build_audiocpp(audiocpp_dir, settings["backend"], emit=emit, cancel=cancel) if rc != 0: print(f"[WARNING] build exited with code {rc}; the server.json " "was still written — build audiocpp_server manually " "before starting it") else: print("[OK] build complete") return rc steps.append(taskview.TaskStep( f"Build audiocpp_server ({settings['backend']})", build)) def transcribe(emit, cancel): args.input_dir = settings["wav_dir"] if settings["include_clone"] and args.input_dir is not None: transcripts, write_prompt = _transcribe( args, True, plan=settings["plan"], cancel=cancel) elif args.input_dir is not None: print(f"[WARNING] Ignoring {args.input_dir}: no clone-capable " "family selected, so voice presets are not used") transcripts, write_prompt = {}, False else: transcripts, write_prompt = {}, False state["transcripts"] = transcripts state["write_prompt"] = write_prompt return 0 steps.append(taskview.TaskStep("Transcribe reference voices", transcribe)) def write(emit, cancel): # Port sync (applied now that the terminal is back). if settings["sync_port"] is True: _apply_port_sync(settings["port"], True) elif settings["sync_port"] is False: _apply_port_sync(settings["port"], False) _write_and_advise( audiocpp_dir, settings["wav_dir"], settings["output_path"], settings["model_entries"], settings["install_guidance"], settings["host"], settings["port"], settings["backend"], settings["lazy_load"], state["transcripts"], state["write_prompt"]) # Delete-unused cleanup (modify flow): remove the already-downloaded # models the new selection dropped. The regenerated server.json # already only lists the kept entries. if settings.get("delete_unused"): removed = delete_model_files(settings["output_path"], settings["unused_entries"]) print(f"[OK] Deleted {removed} unused model " f"{'entry' if removed == 1 else 'entries'} from disk.") if len(settings["entry_ids"]) == 1: _offer_config_model_id_sync(settings["entry_ids"][0], settings["sync_model_ids"]) print_empty_transcript_warning(state["transcripts"]) return 0 steps.append(taskview.TaskStep("Write server.json & sync config", write)) def install(emit, cancel): _install_models(audiocpp_dir, settings["install_guidance"], settings["download"], emit=emit, cancel=cancel) _print_launch_hint(audiocpp_dir, settings["output_path"]) return 0 install_title = "Download models" if settings.get("download") \ else "Print model install commands" steps.append(taskview.TaskStep(install_title, install)) return steps def _execute(settings: dict, args: argparse.Namespace) -> int: """Shared console tail: build, sync, transcribe, write, install, advise. Runs after the TUI wizard returns (or after _collect_from_flags for a non-interactive run): the terminal is plain, so subprocess output and transcription progress appear normally. The same work as ``_execute_steps``, run with no emit (console streaming). """ return taskview.run_steps_inline(_execute_steps(settings, args)) def setup_screen(stdscr) -> int: """Run the setup wizard on an existing curses screen (the hub's). The hub drives this as one screen of its own ``tui.Wizard`` stack, so Esc on the wizard's first screen simply returns here and the hub pops back to the menu that launched it. The setup tail (build/transcribe/ write/download) runs inside the TUI task view on this same screen, so the hub's curses session stays intact and the user sees per-step status and progress instead of being dropped to the console. Returns 0 on completion, 1 when the user aborted. """ parser = build_parser() args = parser.parse_args([]) settings = _wizard(stdscr, args, parser) if settings is None: return 1 return taskview.run_steps(stdscr, "Setting up audio.cpp", _execute_steps(settings, args)) def build_screen(stdscr) -> int: """Build audiocpp_server from the hub when the checkout has no binary. Asks which backend to build for (pre-selecting the backend an existing server.json records, else cuda), runs the build inside the TUI task view, then updates server.json's ``backend`` field to match. Returns 0 on success, non-zero when the user backed out, cancelled, or the build failed. This is the hub's "Build audio.cpp server" action, so a checkout that was cloned but never built is always buildable from the TUI. """ checkout = find_local_checkout() if checkout is None: tui.flash(stdscr, "No audio.cpp checkout found — install audio.cpp " "first.", "err") return 1 if find_audiocpp_server_bin(checkout) is not None: tui.flash(stdscr, "audiocpp_server is already built.", "ok") return 0 server_config = load_server_config(checkout / "server.json") or {} recorded = server_config.get("backend") options, default = _backend_options(None) if recorded in BACKENDS: default = next((i for i, (_label, value) in enumerate(options) if value == recorded), default) backend = tui.menu( stdscr, "Which inference backend should audiocpp_server be built " "for?", options, default_index=default, back_value=_GO_BACK) if backend is _GO_BACK: return 1 rc = taskview.run_steps(stdscr, "Build audiocpp_server", [ taskview.TaskStep( f"Build audiocpp_server ({backend})", lambda emit, cancel: build_audiocpp( checkout, backend, emit=emit, cancel=cancel))]) if rc != 0: return rc if update_server_backend(backend): tui.flash(stdscr, f"audiocpp_server built for {backend}.", "ok") else: tui.flash(stdscr, f"audiocpp_server built for {backend}. (Could not " "update server.json's backend field — reconfigure audio.cpp " "if it was already configured.)", "warn") return 0 def update_server_backend(backend: str) -> bool: """Rewrite the 'backend' in the checkout's server.json, or True when none. Sets ``backend`` to BACKEND in ``/server.json`` (same ``json.dump`` formatting as the wizard). Returns True when the file now carries BACKEND, when there is no server.json (nothing to sync), or when it already does; False when the file exists but cannot be read/written. """ checkout = find_local_checkout() if checkout is None: return True server_json = checkout / "server.json" if not server_json.exists(): return True try: data = json.loads(server_json.read_text(encoding="utf-8")) except (OSError, ValueError): return False if not isinstance(data, dict): return False if data.get("backend") == backend: return True data["backend"] = backend try: with server_json.open("w", encoding="utf-8") as handle: json.dump(data, handle, indent=2, ensure_ascii=False) handle.write("\n") except OSError: return False return True def run_tui(args: Optional[argparse.Namespace] = None, parser: Optional[argparse.ArgumentParser] = None) -> int: """Run the audio.cpp setup wizard end-to-end. With no ARGS (the hub's call) a default namespace is built so the full wizard runs. Called from ``main`` after argparse when the terminal is interactive. Returns the process exit code. """ import curses if args is None: parser = build_parser() args = parser.parse_args([]) if args.input_dir is not None and not args.input_dir.is_dir(): print(f"[ERROR] --wavs not found: {args.input_dir}", file=sys.stderr) return 2 try: settings = curses.wrapper(_wizard, args, parser) except _TuiError as exc: print(f"[ERROR] {exc}", file=sys.stderr) return 2 except tui.WizardCancelled: print("\n[INFO] Cancelled; nothing was written") return 1 try: curses.curs_set(1) # restore the text cursor hidden by the TUI except curses.error: pass if settings is None: print("[INFO] Aborted; existing server.json kept") return 1 return _execute(settings, args) def _collect_from_flags(args: argparse.Namespace, parser: argparse.ArgumentParser) -> Optional[dict]: """Build the settings dict from flags for a non-interactive run. Every required value must come from a flag (there are no prompts in a non-interactive run); a missing one is a hard ``parser.error``. Returns the settings dict, or None when the user declined an overwrite (the default-location fallback then also exists). """ # Checkout: --audiocpp-dir, else a local checkout, else --clone clones one. audiocpp_dir = args.audiocpp_dir if audiocpp_dir is None: audiocpp_dir = find_local_checkout() if audiocpp_dir is None and args.clone: target = APP_DIR / AUDIOCPP_DIR_NAME rc = common.git_clone(AUDIOCPP_GIT_URL, target) if rc != 0: parser.error(f"git clone failed (exit {rc}); clone audio.cpp " f"manually: git clone {AUDIOCPP_GIT_URL} {target}") audiocpp_dir = target if audiocpp_dir is None: parser.error( "An audio.cpp checkout is required. Pass --audiocpp-dir PATH, " "or --clone to clone app/audio.cpp, or run without flags for the " "TUI wizard.") audiocpp_dir = Path(audiocpp_dir).resolve() if not audiocpp_dir.is_dir(): parser.error(f"audio.cpp checkout not found: {audiocpp_dir}") root = _resolve_audiocpp_root(audiocpp_dir) if root is None: parser.error(f"{audiocpp_dir} has no model_specs/ directory; point " "--audiocpp-dir at the root of an audio.cpp checkout") audiocpp_dir = root try: catalog = load_model_catalog(audiocpp_dir) except NotADirectoryError as exc: parser.error(str(exc)) if not catalog: parser.error( f"No TTS model families found in {audiocpp_dir}/model_specs; " "check the checkout is up to date") catalog_by_family = {entry["family"]: entry for entry in catalog} # Families: required from --families in a non-interactive run. if args.families is None: parser.error("--families is required in a non-interactive run (or run " "without flags for the TUI wizard)") requested = [f.strip() for f in args.families.split(",") if f.strip()] unknown = [f for f in requested if f not in catalog_by_family] if unknown: parser.error( f"Unknown family in --families: {', '.join(unknown)}. " f"Available: {', '.join(catalog_by_family)}") family_keys: List[str] = [] for fam in requested: if fam not in family_keys: family_keys.append(fam) chosen: Dict[str, List[dict]] = {} for family in family_keys: opts = package_dir_options(catalog_by_family[family]) if args.all_packages: chosen[family] = opts else: chosen[family] = [opt for opt in opts if opt["recommended"]] # Non-interactive pickers: design packages default to vdes, dup ids get -2. def task_picker(install_id: str) -> str: return TASK_VDES def id_picker(display_name: str, install_id: str, default: str) -> str: return default model_entries, entry_ids, install_guidance, design_entry_ids, include_clone = \ _build_entries(family_keys, chosen, catalog_by_family, task_picker, id_picker) # Server settings. host = args.host or DEFAULT_HOST detected_backend = detect_backend(audiocpp_dir) if args.build_backend: backend = args.build_backend build = detected_backend is None elif args.backend: backend = args.backend build = False elif detected_backend is not None: backend = detected_backend build = False else: backend = "cuda" build = False port = args.port if args.port is not None else config_port() lazy_load = True # Output path / overwrite (decline falls back to cwd, then aborts). output_path = args.output if args.output is not None \ else audiocpp_dir / "server.json" if output_path.exists() and not args.force: if args.output is None: output_path = Path.cwd() / "server.json" if output_path.exists() and not args.force: print("[INFO] Aborted; existing server.json kept") return None else: print("[INFO] Aborted; existing server.json kept") return None # Config sync decisions (auto-apply unless explicitly declined). sync_port: Optional[bool] = None if port != config_port(): sync_port = not args.no_sync_port sync_model_ids: Optional[bool] = None if len(entry_ids) == 1 and not ( config.AUDIOCPP_MODEL_ID == entry_ids[0] and config.AUDIOCPP_CLONE_MODEL_ID == entry_ids[0]): sync_model_ids = not args.no_sync_model_ids # Wav dir + transcription plan (defaults to the project's voices/ dir). wav_dir = args.input_dir if args.input_dir is not None else VOICES_DIR plan: Optional[dict] = None if include_clone and wav_dir is not None: wav_files = find_wav_files(wav_dir) if wav_files: prompt_path = wav_dir / PROMPT_TEXT_FILENAME plan = _flag_plan(wav_files, prompt_path, args.force) return { "audiocpp_dir": audiocpp_dir, "catalog": catalog, "catalog_by_family": catalog_by_family, "output_path": output_path, "family_keys": family_keys, "chosen": chosen, "model_entries": model_entries, "entry_ids": entry_ids, "install_guidance": install_guidance, "design_entry_ids": design_entry_ids, "include_clone": include_clone, "host": host, "port": port, "backend": backend, "build": build, "lazy_load": lazy_load, "sync_port": sync_port, "sync_model_ids": sync_model_ids, "wav_dir": wav_dir, "plan": plan, "download": args.download, } def build_parser() -> argparse.ArgumentParser: """The audio.cpp setup CLI (also used to build a default namespace).""" parser = argparse.ArgumentParser( description="Set up the audio.cpp TTS backend: clone/build, pick " "models, write server.json, and sync app/converter/config.py.") parser.add_argument("--wavs", type=resolve_wav_dir_arg, default=None, dest="input_dir", metavar="WAV_DIR", help="Directory with .wav reference files to publish as " "a server-level voice_dir cloning library " f"(default: {VOICES_DIR}; asked for when omitted " "in the TUI)") parser.add_argument("--output", type=Path, default=None, help="Output path for server.json (default: " "server.json inside the audio.cpp checkout; an " "existing file is overwritten only with --force " "or a TUI confirm)") parser.add_argument("--audiocpp-dir", type=normalize_dir_arg, default=None, help="Path to a local audio.cpp checkout containing a " "model_specs/ directory (default: detected from " "AUDIOCPP_DIR or ./app/audio.cpp; in the TUI you can " "clone one instead)") parser.add_argument("--clone", action="store_true", help="Non-interactive: clone audio.cpp into " "./app/audio.cpp when no checkout is found") parser.add_argument("--families", type=str, default=None, help="Comma-separated model families to host, as named " "in the audio.cpp catalog (e.g. " "qwen3_tts,higgs_audio_tts). Required in a " "non-interactive run; skips the family tree in " "the TUI") parser.add_argument("--all-packages", action="store_true", help="Host every installable package of each selected " "family (distinct target_directory) instead of " "only the recommended one. Voice-design packages " "are hosted with task 'vdes'") parser.add_argument("--host", type=str, default=None, help="Bind host for the server (default: 127.0.0.1)") parser.add_argument("--port", type=int, default=None, help="Port for the server (default: the port in " "AUDIOCPP_API_URL from app/converter/config.py)") parser.add_argument("--backend", choices=BACKENDS, default=None, help="Inference backend recorded in server.json " "(default: auto-detected from the checkout's " "build/ directory, else cuda)") parser.add_argument("--build-backend", choices=BACKENDS, default=None, help="Build audiocpp_server for this backend when it " "is not built yet, and use it in server.json") parser.add_argument("--whisper-model", type=str, default="base", help="Whisper model size for transcription " "(default: base)") parser.add_argument("--force", action="store_true", help="Overwrite the output file (and prompt_text) " "without prompting; in the TUI, start the " "wizard fresh instead of loading the existing " "server.json") parser.add_argument("--download", action="store_true", help="Run model_manager_v2.py install for each hosted " "model automatically (default: print the commands " "only)") parser.add_argument("--no-sync-port", action="store_true", help="Do not rewrite AUDIOCPP_API_URL in " "app/converter/config.py when --port differs") parser.add_argument("--no-sync-model-ids", action="store_true", help="Do not rewrite AUDIOCPP_MODEL_ID/" "AUDIOCPP_CLONE_MODEL_ID for a single-entry server") return parser def detect() -> BackendStatus: """Detect how far audio.cpp is set up, plus the command to start it.""" checkout = find_local_checkout() details: List[str] = [] launch = "" if checkout is None: # No local checkout: only a remote server can make this usable. remote = _detect_remote() return BackendStatus("audiocpp", "audio.cpp", installed=False, configured=False, running=remote[0], remote=remote[0], remote_urls=remote[1], details=["not cloned — run setup to clone " "./app/audio.cpp"]) details.append(f"checkout: {checkout}") binary = find_audiocpp_server_bin(checkout) built = binary is not None if built: details.append(f"built: {binary}") else: details.append("not built — run setup to build audiocpp_server") server_json = checkout / "server.json" configured = server_json.exists() specs: List[ServerSpec] = [] missing = missing_model_entries(server_json) if configured else [] if configured: details.append(f"config: {server_json}") if missing: # The config references model files that are not on disk; a # conversion would fail at model-load time, so say so now. details.extend(model_install_hints(checkout, missing)) if built: # Spawned from the checkout: audiocpp_server discovers # model_specs/.json relative to its working directory. specs = [ServerSpec( "audiocpp", config.AUDIOCPP_API_URL, [str(binary), "--config", str(server_json)], cwd=checkout, identity=probe.IDENTITY_AUDIOCPP)] else: launch = (f"cd {checkout} && ./build/--release" f"/bin/audiocpp_server --config {server_json}") else: details.append("no server.json — run setup to configure models") if specs: launch = format_launch_hint(specs) managed = servers.manages(specs) remote_running, remote_urls = _detect_remote(managed) # A more specific "part-way set up" label than unavailable/installed: # cloned but never built, or built but not configured. partial = "" if not built: partial = "downloaded (not built)" elif not configured: partial = "built (not configured)" return BackendStatus("audiocpp", "audio.cpp", installed=built, configured=configured, running=managed or remote_running, details=details, launch_hint=launch, servers=specs, managed=managed, remote=remote_running, remote_urls=remote_urls, models_missing=bool(missing), partial=partial) def _detect_remote(managed: bool = False) -> Tuple[bool, dict]: """Detect an externally-run audiocpp_server at the remote URL. Returns ``(running, {spec_name: url})``. The remote URL is probed only when configured (non-empty); a server answering there is ignored when it is this tool's own managed server (remote URL == local URL and our pid is still alive) — that instance is already reported as "[local]". """ url = (config.AUDIOCPP_REMOTE_URL or "").strip() if not url: return False, {} if managed and probe.same_endpoint(url, config.AUDIOCPP_API_URL): return False, {} if probe.identify_server(url) == probe.IDENTITY_AUDIOCPP: return True, {"audiocpp": url} return False, {} def main() -> int: parser = build_parser() args = parser.parse_args() if args.input_dir is not None and not args.input_dir.is_dir(): parser.error( f"WAV directory not found: {args.input_dir}\n" f" (resolved from the current working directory: " f"{Path.cwd()})\n" " --wavs must be a directory containing the .wav " "reference files to use as voice cloning presets") if _interactive(): return run_tui(args, parser) # Non-interactive (no terminal, or all flags supplied): flag-only path. settings = _collect_from_flags(args, parser) if settings is None: return 1 return _execute(settings, args) if __name__ == "__main__": sys.exit(main())