diff options
| author | historia <historiavg@proton.me> | 2026-08-26 02:25:55 -0400 |
|---|---|---|
| committer | historia <historiavg@proton.me> | 2026-08-26 02:25:55 -0400 |
| commit | 8b5c8697740ff415cf7f1d03c9fb5a8c8851d420 (patch) | |
| tree | 28c0323c54c896af5f89fb34b89a62e0fe0df291 /app/backends/audiocpp.py | |
| parent | acbd9ff2c91182d96c57ffb57bee6e9b3fcbcbd4 (diff) | |
| download | tts-audiobook-generator-8b5c8697740ff415cf7f1d03c9fb5a8c8851d420.tar.gz | |
refactor: audiocpp.py setup flow
Diffstat (limited to 'app/backends/audiocpp.py')
| -rwxr-xr-x | app/backends/audiocpp.py | 2556 |
1 files changed, 0 insertions, 2556 deletions
diff --git a/app/backends/audiocpp.py b/app/backends/audiocpp.py deleted file mode 100755 index 3cca94c..0000000 --- a/app/backends/audiocpp.py +++ /dev/null @@ -1,2556 +0,0 @@ -#!/usr/bin/env python3 -"""Set up the audio.cpp TTS backend for the audiobook generator. - -This does the whole audio.cpp setup end-to-end as a full-screen DOS-style -TUI: locate or clone an audio.cpp checkout into ``app/audio.cpp``, optionally -build ``audiocpp_server``, pick model families/packages from the checkout's -``model_specs`` catalog, transcribe reference .wav voices, write -``server.json``, sync ``app/converter/config.py``, download the models, and -print the exact command to start the server. It is driven by -``audiobook.py``'s TUI hub (``backends.REGISTRY``) but can also be run -directly for scripting — every value has a flag, and a non-interactive run -with all flags supplied never opens the TUI. - -The converter is family-agnostic (it detects the family of the selected -entry from ``GET /v1/models`` at startup), so any TTS family listed in the -catalog works without further changes. - -Usage: - python app/backends/audiocpp.py [--wavs WAV_DIR] [--output PATH] - [--clone] [--families FAM1,FAM2] - [--all-packages] [--host HOST] [--port PORT] - [--build-backend {cuda,vulkan,hip,cpu}] [--backend {cuda,vulkan,hip,cpu}] - [--whisper-model NAME] [--force] - [--download] [--no-sync-port] [--no-sync-model-ids] - -With no flags and a terminal, the TUI wizard runs. Without a terminal -(or with all flags supplied), it runs non-interactively from the flags; -any missing required value is a hard error with a remediation hint. - -When the target ``server.json`` already exists, the TUI wizard runs as a -"modify": it loads the existing models, host, port, backend and voice -directory and pre-fills the screens with them (the model tree -opens with the installed models already checked) instead of prompting to -overwrite, and offers to delete already-downloaded models that are no -longer selected. -""" - -import argparse -import contextlib -import io -import json -import os -import re -import shlex -import shutil -import sys -import tempfile -import urllib.parse -import urllib.request -from datetime import datetime -from pathlib import Path -from typing import Callable, Dict, List, Optional, Set, Tuple - -# Allow running directly (python app/backends/audiocpp.py) from any cwd. -sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) - -from backends import ( - BackendStatus, - ServerSpec, - common, - format_launch_hint, - probe, - servers, -) -from backends.common import ( - APP_DIR, - CONFIG_PATH, - PROMPT_TEXT_FILENAME, - TTS_ROOT, - VOICES_DIR, - detect_wav_dir, - find_wav_files, - read_prompt_text, - resolve_wav_dir_arg, - url_with_port, - write_prompt_text, -) -from backends.common import ( - wav_dir_info as _wav_dir_info, -) -from backends.common import ( - wav_dir_preview as _wav_dir_preview, -) -from converter import config -from converter.clients import transcribe_reference_audio, whisper_backend_available -from ui import taskview, tui - -DEFAULT_HOST = "127.0.0.1" -FALLBACK_PORT = 8080 - -BACKENDS = ("cuda", "vulkan", "hip", "cpu") - -TASK_TTS = "tts" -TASK_VDES = "vdes" - -# audio.cpp is cloned into the app directory of the audiobook generator. -AUDIOCPP_DIR_NAME = "audio.cpp" -AUDIOCPP_GIT_URL = "https://github.com/0xShug0/audio.cpp" - -# ggml build patches shipped in this repo and applied to the (gitignored) -# audio.cpp checkout before building, so a fresh clone survives known ggml -# build bugs the audio.cpp fork has not re-vendored yet. See -# apply_ggml_patches() below. -PATCH_DIR = Path(__file__).resolve().parent / "patches" - -# Sentinel returned by tui.confirm (via its cancel_value) when the user -# presses Esc on an overwrite prompt to go back to the wav-directory browser -# instead of aborting the wizard. -_GO_BACK = object() - - -class _GoBack(Exception): - """Internal signal: Esc was pressed inside one of a screen's sub-prompts. - - The wizard drives a stack of screens via ``tui.Wizard``. Helpers that ask - several questions through callbacks (the task/id pickers inside - ``_build_entries``, the transcription plan, the download prompt) cannot - themselves return the wizard's ``BACK`` sentinel, so they convert the - ``_GO_BACK`` value passed to each widget into this exception. The screen - that invoked the helper catches it and returns ``tui.Wizard.BACK``, which - pops back to the previous screen. Esc on the first screen aborts the - whole wizard. - """ - -# Package names that mark a voice-design model (hosted with task "vdes"). -DESIGN_PACKAGE_RE = re.compile(r"voice[\s_\-]?design", re.IGNORECASE) - - -class _TuiError(Exception): - """A fatal error raised from inside the TUI wizard. - - The message is reported to stderr after the terminal is restored; the - process exits with code 2 (matching a parser error). - """ - - -def _interactive() -> bool: - """True when the TUI wizard can run (curses importable + tty).""" - try: - import curses # noqa: F401 - except ImportError: - return False - try: - return sys.stdin.isatty() and sys.stdout.isatty() - except (AttributeError, ValueError): - return False - - -# Backend display order, with short descriptions. The backend name is padded -# so the descriptions' dashes line up in the menu. -_BACKEND_DESCRIPTIONS = ( - ("cuda", "NVIDIA GPUs (fastest)"), - ("vulkan", "cross-vendor GPU"), - ("hip", "AMD GPUs"), - ("cpu", "no GPU required"), -) - - -def _backend_options(detected: Optional[str] = None - ) -> Tuple[List[Tuple[str, str]], int]: - """Build the aligned backend menu options and the default index. - - The backend names are padded to a common width so the ``-`` dashes - before the descriptions line up. When DETECTED matches one of the - options, that option gets ``[auto-detected]`` appended and is the - default (cursor/start) selection; otherwise the first option is the - default as before. Returns (options, default_index). - """ - width = max(len(name) for name, _ in _BACKEND_DESCRIPTIONS) - options: List[Tuple[str, str]] = [] - default_index = 0 - for index, (name, desc) in enumerate(_BACKEND_DESCRIPTIONS): - label = f"{name.ljust(width)} - {desc}" - if detected == name: - label += " [auto-detected]" - default_index = index - options.append((label, name)) - return options, default_index - - -def config_port() -> int: - """Return the port of AUDIOCPP_API_URL in app/converter/config.py.""" - try: - return urllib.parse.urlsplit(config.AUDIOCPP_API_URL).port or FALLBACK_PORT - except ValueError: - return FALLBACK_PORT - - -def update_config_api_url_port(port: int, config_path: Optional[Path] = None) -> bool: - """Rewrite the port inside AUDIOCPP_API_URL in app/converter/config.py. - - Reads the configured URL from the file (not from the imported module, - which a long hub session can leave behind), swaps its port for PORT, - and writes it back through ``common.update_config_value`` so the - imported module mirrors the change immediately. Returns True when the - file now holds the new URL. - """ - path = Path(config_path) if config_path is not None else CONFIG_PATH - try: - text = path.read_text(encoding="utf-8") - except OSError: - return False - match = re.search(r'(?m)^\s*AUDIOCPP_API_URL\s*=\s*"([^"]*)"', text) - if not match: - return False - return common.update_config_value("AUDIOCPP_API_URL", - url_with_port(match.group(1), port), - config_path=path) - - -def update_server_config_port(port: int) -> bool: - """Rewrite the 'port' in the audio.cpp checkout's server.json. - - Loads ``<checkout>/server.json``, sets its ``port`` to PORT, and - rewrites it with the same ``json.dump`` formatting the wizard uses. - Returns True when the file now carries PORT (a no-op when it already - does), and False when there is no checkout/server.json or the file - cannot be read or written. - """ - checkout = find_local_checkout() - if checkout is None: - return False - server_json = checkout / "server.json" - if not server_json.exists(): - return False - try: - data = json.loads(server_json.read_text(encoding="utf-8")) - except (OSError, ValueError): - return False - if not isinstance(data, dict): - return False - if data.get("port") == port: - return True - data["port"] = port - try: - with server_json.open("w", encoding="utf-8") as handle: - json.dump(data, handle, indent=2, ensure_ascii=False) - handle.write("\n") - except OSError: - return False - return True - - -def update_config_model_ids(model_id: str, - clone_model_id: Optional[str] = None, - config_path: Optional[Path] = None) -> bool: - """Rewrite AUDIOCPP_MODEL_ID (and AUDIOCPP_CLONE_MODEL_ID when given). - - Goes through ``common.update_config_value`` so the imported config - module mirrors the change immediately. Returns True when every named - key now holds its value in the file. - """ - path = Path(config_path) if config_path is not None else CONFIG_PATH - ok = common.update_config_value("AUDIOCPP_MODEL_ID", model_id, - config_path=path) - if clone_model_id is not None: - ok = common.update_config_value("AUDIOCPP_CLONE_MODEL_ID", - clone_model_id, - config_path=path) and ok - return ok - - -# audio.cpp build directories are named ``<platform>-<backend>-<type>`` (e.g. -# ``linux-cuda-release``, ``windows-vulkan-debug``, ``macos-metal-release``) -# and the built server lands in ``<that>/bin/audiocpp_server``. The Metal -# macOS backend is reported as "cpu" here since it is not a separate -# --backend choice for audiocpp_server. -_BACKEND_TOKEN_RE = re.compile(r"-(cuda|vulkan|hip|cpu|metal)(?:-|$)") - - -def detect_backend(audiocpp_dir: Path) -> Optional[str]: - """Best-effort detection of the backend audiocpp_server was built for. - - Scans ``audiocpp_dir/build/*`` for build directories that contain a - built ``bin/audiocpp_server`` (``.exe`` allowed on Windows) and reads - the backend token out of the directory name (``-cuda-``, ``-vulkan-``, - ``-hip-`` or ``-cpu-``; ``-metal-`` is mapped to ``cpu``). Returns the - backend only when exactly one distinct backend was built, so a checkout - with builds for several backends does not silently pick one. Returns - None when there is no ``build/`` directory, no built server, or more - than one distinct backend. - """ - build_root = audiocpp_dir / "build" - if not build_root.is_dir(): - return None - backends: Set[str] = set() - try: - build_dirs = sorted(build_root.iterdir(), - key=lambda p: p.name.lower()) - except OSError: - return None - for build_dir in build_dirs: - if not build_dir.is_dir(): - continue - server = build_dir / "bin" / "audiocpp_server" - if not server.exists(): - server_exe = build_dir / "bin" / "audiocpp_server.exe" - if not server_exe.exists(): - continue - match = _BACKEND_TOKEN_RE.search(build_dir.name.lower()) - if not match: - continue - token = match.group(1) - backends.add("cpu" if token == "metal" else token) - if len(backends) == 1: - return next(iter(backends)) - return None - - -def _default_package(packages: List[dict]) -> Optional[dict]: - """Pick the default package from a list of packages. - - Prefers the package flagged ``default: true``, then the first GGUF - package, then the first package overall. Returns None for an empty list. - """ - if not packages: - return None - for package in packages: - if package.get("default"): - return package - for package in packages: - if package.get("format") == "gguf": - return package - return packages[0] - - -def load_model_catalog(audiocpp_dir: Path) -> List[dict]: - """Read model_specs/*.json and return the TTS-capable families. - - Each returned entry has: family, display_name, description, languages, - clone_capable, packages (the full list from the spec), install_id - (recommended package id), and default_path (``models/<target_directory>``). - All families are treated equally and listed in alphabetical order by - display name. - """ - specs_dir = audiocpp_dir / "model_specs" - if not specs_dir.is_dir(): - raise NotADirectoryError( - f"{audiocpp_dir} has no model_specs/ directory; re-run setup " - "to refresh the audio.cpp checkout") - entries: List[dict] = [] - for spec_path in sorted(specs_dir.glob("*.json")): - try: - spec = json.loads(spec_path.read_text(encoding="utf-8")) - except (OSError, ValueError): - continue - tasks = spec.get("tasks") or [] - if "tts" not in tasks and spec.get("category") != "tts": - continue - family = spec.get("family") or spec_path.stem - packages = spec.get("packages") or [] - package = _default_package(packages) - if package is None: - # No installable package: skip (cannot be hosted from a path). - continue - target_directory = package.get("target_directory") or family - languages = spec.get("languages") or [] - display_name = spec.get("display_name") or family - description = spec.get("description") or "" - entries.append({ - "family": family, - "display_name": display_name, - "description": description, - "languages": languages, - "tasks": list(tasks), - "clone_capable": "clone" in tasks, - "packages": packages, - "install_id": package.get("id") or family, - "default_path": f"models/{target_directory}", - }) - - # All families are treated equally: alphabetical by display name. - entries.sort(key=lambda entry: entry["display_name"].lower()) - return entries - - -def is_design_package(package: dict) -> bool: - """Return True when a package's name marks it a voice-design model. - - audio.cpp voice-design packages (whose id, display name, or target - directory mentions "voice design") are the only packages that must be - hosted with task "vdes"; their role is not in the schema, only in those - strings, so it is detected from them. - """ - text = " ".join(str(package.get(key, "")) - for key in ("id", "display_name", "target_directory")) - return bool(DESIGN_PACKAGE_RE.search(text)) - - -def package_dir_options(entry: dict) -> List[dict]: - """Return one option per distinct target_directory of a family's packages. - - Each option is a dict with: target_directory, install_id (the recommended - package id inside that directory), design (voice-design package flag), and - recommended (whether it holds the family's default package). Precisions - that share a directory (q8_0/bf16/...) collapse to a single option. - """ - packages = entry.get("packages") or [] - default_pkg = _default_package(packages) - default_dir = (default_pkg or {}).get("target_directory") or entry["family"] - by_dir: Dict[str, List[dict]] = {} - order: List[str] = [] - for package in packages: - directory = package.get("target_directory") or entry["family"] - if directory not in by_dir: - by_dir[directory] = [] - order.append(directory) - by_dir[directory].append(package) - options: List[dict] = [] - for directory in order: - package = _default_package(by_dir[directory]) - options.append({ - "target_directory": directory, - "install_id": (package or {}).get("id") or directory, - "design": is_design_package(package or {}), - "recommended": directory == default_dir, - }) - # Put the recommended package first for a friendlier checklist. - options.sort(key=lambda opt: not opt["recommended"]) - return options - - -def build_model_entry(family: str, model_id: str, model_path: str, - task: str = TASK_TTS) -> dict: - """Assemble one server.json model entry. - - ``task`` defaults to "tts"; voice design packages are hosted with - "vdes" so the server runs its design session for speech requests - (audiobook.py then requires --instructions with that entry). - """ - return { - "id": model_id, - "family": family, - "path": model_path, - "task": task, - "mode": "offline", - } - - -def build_server_config(host: str, port: int, backend: str, lazy_load: bool, - model_entries: List[dict], - voice_dir: Optional[str] = None) -> dict: - """Assemble the server.json document. - - ``voice_dir`` is a server-level cloning voice library; when set, every - hosted clone-capable family can use its voices with ``--voice``. - """ - config_doc = { - "host": host, - "port": port, - "backend": backend, - "lazy_load": lazy_load, - "models": model_entries, - } - if voice_dir: - config_doc["voice_dir"] = voice_dir - return config_doc - - -def transcribe_wav_dir(wav_files: list, whisper_model: str, - cancel=None) -> Dict[str, str]: - """Transcribe each wav file and return a mapping of stem -> transcript. - - CANCEL (a ``threading.Event``) is checked between files so the in-TUI - task view can stop a long transcription early. - """ - transcripts: Dict[str, str] = {} - for wav_file in wav_files: - if cancel is not None and cancel.is_set(): - print("[INFO] Transcription cancelled") - break - name = wav_file.stem - print(f"[INFO] Transcribing {wav_file.name} (voice '{name}')...") - text = transcribe_reference_audio(str(wav_file), model_name=whisper_model) - if text: - print(f"[OK] {name}: {text}") - else: - print(f"[WARNING] No transcript for '{name}'; cloning works best " - "with an accurate transcript — consider editing prompt_text " - "by hand before starting the server") - transcripts[name] = text or "" - return transcripts - - -def print_empty_transcript_warning(transcripts: Dict[str, str]) -> None: - """Print a loud, final warning for voices whose transcript is empty.""" - empty = sorted(name for name, text in transcripts.items() if not text) - if not empty: - return - bar = "=" * 70 - print() - print(bar) - print("[WARNING] MANUAL TRANSCRIPTION REQUIRED") - print(bar) - listing = " - " + "\n - ".join(empty) if len(empty) > 1 else f" - {empty[0]}" - print(f"The following voice(s) have an EMPTY transcript in prompt_text:\n" - f"{listing}") - print("Those voices will NOT work until you add an accurate transcript.") - print(f"Edit {PROMPT_TEXT_FILENAME} in your voice directory and fill in the " - "text after '|' for each voice above.") - print(bar) - - -def _apply_port_sync(port: int, accepted: bool) -> None: - """Write the port into app/converter/config.py, or report when declined.""" - if accepted: - if not update_config_api_url_port(port): - print(f"[WARNING] Could not update {CONFIG_PATH}; edit " - "AUDIOCPP_API_URL by hand so audiobook.py uses the " - "new port") - else: - print("[WARNING] Left AUDIOCPP_API_URL unchanged; audiobook.py " - f"will still use port {config_port()}") - - -def _decide_transcription(wav_files: list, existing: Dict[str, str], - prompt_exists: bool, force: bool, - confirm: Callable[[str, bool], bool]) -> dict: - """Decide which voices to transcribe; CONFIRM asks the plan questions. - - Returns a plan dict: {"mode": "all"|"missing"|"keep", "missing": - [...], "existing": {...}} — "existing" carries the prompt_text - mapping read while deciding, so the caller can reuse it instead of - reading the file again. - """ - mode = "all" - missing: List[Path] = [] - if prompt_exists and not force: - missing = [wav for wav in wav_files - if not existing.get(wav.stem, "").strip()] - if not missing: - if confirm("All voices already transcribed in prompt_text. " - "Re-transcribe anyway?", False): - mode = "all" - else: - mode = "keep" - elif confirm("Existing transcription and new .wavs detected, " - "only transcribe new voices?", True): - mode = "missing" - else: - mode = "all" - return {"mode": mode, "missing": missing, "existing": existing} - - -def _transcribe(args: argparse.Namespace, plan: Optional[dict], - cancel=None) -> Tuple[Dict[str, str], bool]: - """Transcribe the wav directory into a stem -> transcript mapping. - - Returns the mapping and a flag indicating whether it should be written to - prompt_text (False when an existing, complete prompt_text is kept as-is). - PLAN is always pre-collected — by the TUI (via _decide_transcription and - its confirm callbacks) or by _flag_plan for a non-interactive run — so no - questions are asked here; a None PLAN defaults to "transcribe everything". - CANCEL is checked between files. - """ - wav_files = find_wav_files(args.input_dir) - if not wav_files: - print(f"[WARNING] No .wav files found in {args.input_dir}; writing the " - "config without a voice_dir") - return {}, False - - prompt_path = args.input_dir / PROMPT_TEXT_FILENAME - existing = dict((plan or {}).get("existing") or {}) - mode = plan["mode"] if plan else "all" - - if mode == "keep": - print(f"[INFO] Kept existing {prompt_path}; all voices were " - "already transcribed, nothing new to transcribe") - return existing, False - - if whisper_backend_available() is None: - print("[WARNING] Neither faster_whisper nor whisper was found, so " - "reference .wav files cannot be transcribed automatically and " - "every transcript will be empty.") - print(" Install whisper (or faster_whisper) in your " - "audiobook environment to transcribe automatically; otherwise " - "transcripts must be added by hand (see the warning at the end).") - - if plan["mode"] == "missing": - new_transcripts = transcribe_wav_dir(plan["missing"], args.whisper_model, - cancel=cancel) - transcripts = dict(existing) - transcripts.update(new_transcripts) - else: - transcripts = transcribe_wav_dir(wav_files, args.whisper_model, - cancel=cancel) - return transcripts, True - - -def _flag_plan(wav_files: list, prompt_path: Path, force: bool) -> dict: - """Build a transcription plan for a non-interactive (flag-only) run. - - With --force everything is re-transcribed; otherwise an existing - prompt_text is reused and only voices with an empty transcript are - re-transcribed, mirroring what the TUI confirms interactively. - """ - if prompt_path.exists() and not force: - existing = read_prompt_text(prompt_path) - missing = [wav for wav in wav_files - if not existing.get(wav.stem, "").strip()] - if not missing: - return {"mode": "keep", "missing": [], "existing": existing} - return {"mode": "missing", "missing": missing, "existing": existing} - return {"mode": "all", "missing": [], "existing": {}} - - -def _offer_config_model_id_sync(model_id: str, accepted: Optional[bool]) -> None: - """Point app/converter/config.py at a single hosted model entry. - - The converter requests the model id configured in AUDIOCPP_MODEL_ID, - and single-model servers use the same id for the clone entry, so both - ids are rewritten together. ACCEPTED is True/False (apply/skip the - rewrite) or None when no single-entry sync applies (nothing to do). - """ - if config.AUDIOCPP_MODEL_ID == model_id \ - and config.AUDIOCPP_CLONE_MODEL_ID == model_id: - return - if accepted is None: - return - if accepted: - if not update_config_model_ids(model_id, model_id): - print(f"[WARNING] Could not update {CONFIG_PATH}; edit " - "AUDIOCPP_MODEL_ID and AUDIOCPP_CLONE_MODEL_ID by hand so " - "audiobook.py uses this model") - else: - print("[WARNING] Left the model ids unchanged; audiobook.py will " - f"still request model '{config.AUDIOCPP_MODEL_ID}'") - - -def _build_entries(family_keys: List[str], chosen: Dict[str, List[dict]], - catalog_by_family: Dict[str, dict], - task_picker: Callable[[str], str], - known_tasks: Optional[Dict[Tuple[str, str], str]] = None - ) -> Tuple[List[dict], List[str], List[Tuple[str, str]], - List[str], bool]: - """Build server.json model entries from the selected families/packages. - - TASK_PICKER is called for each design package to choose vdes/tts. - KNOWN_TASKS maps ``(family, target_directory)`` to a previously-stored - task ("tts" or "vdes") so a modify run preserves how a design package - was hosted instead of re-asking. Each entry's server id is its package - ``target_directory`` (flattened to a token), so packages from the same - family never collide; an id that does collide (across families) is - auto-suffixed without prompting. Returns (model_entries, entry_ids, - install_guidance, design_entry_ids, include_clone). - """ - model_entries: List[dict] = [] - entry_ids: List[str] = [] - install_guidance: List[Tuple[str, str]] = [] - design_entry_ids: List[str] = [] - include_clone = False - for family in family_keys: - entry = catalog_by_family[family] - include_clone = include_clone or entry["clone_capable"] - for opt in chosen[family]: - if opt["design"]: - task = known_tasks.get((family, opt["target_directory"])) \ - if known_tasks else None - if task is None: - task = task_picker(opt["install_id"]) - else: - task = TASK_TTS - base_id = opt["target_directory"].replace("/", "-") - model_id = base_id - if model_id in entry_ids: - n = 2 - while f"{base_id}-{n}" in entry_ids: - n += 1 - model_id = f"{base_id}-{n}" - entry_ids.append(model_id) - model_entries.append(build_model_entry( - family, model_id, f"models/{opt['target_directory']}", - task=task)) - install_guidance.append((entry["display_name"], opt["install_id"])) - if task == TASK_VDES: - design_entry_ids.append(model_id) - return (model_entries, entry_ids, install_guidance, - design_entry_ids, include_clone) - - -def _write_and_advise(audiocpp_dir: Path, wav_dir: Optional[Path], - output_path: Path, model_entries: List[dict], - install_guidance: List[Tuple[str, str]], host: str, - port: int, backend: str, lazy_load: bool, - transcripts: Dict[str, str], write_prompt: bool) -> None: - """Console phase shared by both UI modes: write files, print summary. - - After a successful run the console output is the path of the written - server.json. The model install commands (and optional automatic - download) are handled separately by _install_models, called by both - UI modes once the user has decided whether to download. - """ - voice_dir: Optional[str] = None - if transcripts: - if write_prompt: - prompt_path = wav_dir / PROMPT_TEXT_FILENAME - write_prompt_text(wav_dir, transcripts) - print(f"[OK] Wrote {prompt_path}") - voice_dir = str(wav_dir.resolve()) - - server_config = build_server_config( - host=host, port=port, backend=backend, lazy_load=lazy_load, - model_entries=model_entries, voice_dir=voice_dir) - - with output_path.open("w", encoding="utf-8") as handle: - json.dump(server_config, handle, indent=2, ensure_ascii=False) - handle.write("\n") - - count = len(model_entries) - print(f"Wrote {output_path.resolve()} with {count} " - f"{'entry' if count == 1 else 'entries'}.") - - -def _install_models(audiocpp_dir: Path, - install_guidance: List[Tuple[str, str]], - download: bool, emit=None, cancel=None) -> int: - """Print and optionally run the model install commands. - - One ``python <manager> install <id>`` command per hosted model (de-duped - by install id). When DOWNLOAD is True each command is run in the audio.cpp - checkout via ``subprocess`` so the models are downloaded automatically; - a failing install is reported as a warning and does not abort the - remaining downloads. When DOWNLOAD is False (or the model manager is - missing) the commands are only printed, copy-pasteable as before. - - With EMIT given (the in-TUI task view) each download streams its output - to EMIT and — when the checkout's ``model_manager_v2.py`` supports it — - runs with ``--progress --cancel-file`` so the view can show a real byte - progress bar and cancel gracefully. CANCEL aborts a running download. - - Returns 0 when every command succeeded (or nothing needed running), - 130 when cancelled, 1 when any download failed. - """ - manager = audiocpp_dir / "tools" / "model_manager_v2.py" - seen: Set[str] = set() - install_ids: List[str] = [] - for _, install_id in install_guidance: - if install_id not in seen: - seen.add(install_id) - install_ids.append(install_id) - - supports_progress = emit is not None and _manager_supports_progress(manager) - - if download and not manager.is_file(): - print(f"[WARNING] {manager} not found; printing the install commands " - "instead of running them") - download = False - - failed = False - for install_id in install_ids: - command = f"python {manager} install {install_id}" - if not download: - print(command) - continue - print(f"[INFO] Downloading {install_id}...") - argv = [sys.executable, str(manager), "install", install_id] - cancel_file: Optional[Path] = None - on_cancel = None - if supports_progress: - fd, cancel_path = tempfile.mkstemp( - prefix="audiocpp_cancel_", suffix=".cancel") - os.close(fd) - cancel_file = Path(cancel_path) - cancel_file.unlink() # absent = not cancelled - argv += ["--progress", "--cancel-file", str(cancel_file)] - on_cancel = cancel_file.touch - try: - rc = common.run_console_subprocess( - argv, cwd=str(audiocpp_dir), emit=emit, cancel=cancel, - on_cancel=on_cancel) - except OSError as exc: - print(f"[WARNING] Could not run {command}: {exc}") - rc = 1 - finally: - if cancel_file is not None: - try: - cancel_file.unlink() - except OSError: - pass - if rc == 130 or (cancel is not None and cancel.is_set()): - return 130 - if rc != 0: - failed = True - print(f"[WARNING] install {install_id} exited with code " - f"{rc}; the model may need to be downloaded " - "by hand") - return 1 if failed else 0 - - -def _manager_supports_progress(manager: Path) -> bool: - """True when MANAGER (model_manager_v2.py) supports --progress output. - - The ``--progress``/``--cancel-file`` flags are relatively recent; an - older audio.cpp checkout may not have them, so probe the script source - once instead of failing the download with an unknown flag. - """ - try: - text = manager.read_text(encoding="utf-8", errors="ignore") - except OSError: - return False - return "AUDIOCPP_PROGRESS" in text and "--cancel-file" in text - - -def _decide_download(audiocpp_dir: Path, - model_entries: List[dict], - confirm: Callable[[str, bool], bool]) -> bool: - """Ask whether to download the selected models now. - - CONFIRM asks the yes/no question (ask_bool for the line prompts, a TUI - confirm for the wizard). When the audio.cpp model manager is missing the - prompt is skipped and False is returned, so the install commands are only - printed rather than offered to run. The prompt is also skipped (False) - when every selected model is already on disk (see ``_all_models_present``), - so an already-configured checkout is not asked to re-download models it - already has. - """ - manager = audiocpp_dir / "tools" / "model_manager_v2.py" - if not manager.is_file(): - return False - if _all_models_present(audiocpp_dir, model_entries): - return False - return confirm( - "Automatically download the selected models with model_manager_v2.py " - "now?", True) - - -def _build_tree_families(catalog: List[dict]) -> List[dict]: - """Shape the catalog into the checkbox_tree widget's family list.""" - families: List[dict] = [] - for entry in catalog: - capabilities = ["tts"] - if "clone" in entry["tasks"]: - capabilities.append("cloning") - if "design" in entry["tasks"]: - capabilities.append("design") - name = entry["display_name"] - options = [] - for opt in package_dir_options(entry): - options.append({ - "key": opt["target_directory"], - "label": opt["install_id"], - "recommended": opt["recommended"], - }) - families.append({ - "label": name, - "detail": ", ".join(capabilities), - "options": options, - }) - return families - - -def _wizard(stdscr, args: argparse.Namespace, parser: argparse.ArgumentParser - ) -> Optional[dict]: - """Run every TUI screen; return the collected settings, or None to abort. - - The wizard is driven by ``tui.Wizard`` as a stack of screen closures: - each screen shows one interactive widget and returns the next screen - (a closure), ``Wizard.BACK`` (Esc/q pressed — pop to the previous - screen), or the final settings dict. Only screens that actually render - are pushed, so Esc always lands on the previous real screen. A step - whose value is already provided by a flag (``--host``, ``--port``, - ``--families``, ...) or does not apply (e.g. the port-sync prompt when - the port did not change) is folded into the ``_after_*`` guards and - never becomes a screen. Esc on the first screen aborts the whole - wizard. - """ - - s: dict = {} - - def ask_confirm(question: str, default: bool) -> bool: - result = tui.confirm(stdscr, question, default=default, - cancel_value=_GO_BACK) - if result is _GO_BACK: - raise _GoBack() - return result - - def resolve_checkout(audiocpp_dir: Path) -> None: - """Validate the audio.cpp checkout and populate the wizard state ``s``.""" - audiocpp_dir = Path(audiocpp_dir).resolve() - try: - catalog = load_model_catalog(audiocpp_dir) - except NotADirectoryError as exc: - raise _TuiError(str(exc)) - if not catalog: - raise _TuiError(f"No TTS model families found in " - f"{audiocpp_dir}/model_specs; check the " - "checkout is up to date") - catalog_by_family = {entry["family"]: entry for entry in catalog} - output_path = args.output if args.output is not None \ - else audiocpp_dir / "server.json" - # Modify flow: an existing server.json seeds the wizard's screens - # instead of being overwritten from scratch (an explicit --force - # still starts fresh). - existing_config = load_server_config(output_path) \ - if not args.force else None - if existing_config is not None: - existing_selected, existing_tasks = \ - server_config_selections(existing_config, catalog) - else: - existing_selected, existing_tasks = {}, {} - s.update({ - "audiocpp_dir": audiocpp_dir, - "catalog": catalog, - "catalog_by_family": catalog_by_family, - "output_path": output_path, - "existing_config": existing_config, - "existing_selected": existing_selected, - "existing_tasks": existing_tasks, - "existing_host": existing_config.get("host") - if existing_config else None, - "existing_port": existing_config.get("port") - if existing_config else None, - "existing_backend": existing_config.get("backend") - if existing_config else None, - "existing_voice_dir": existing_config.get("voice_dir") - if existing_config else None, - "detected_backend": detect_backend(audiocpp_dir), - }) - - def _families_from_flag() -> None: - requested = [f.strip() for f in args.families.split(",") if f.strip()] - unknown = [f for f in requested if f not in s["catalog_by_family"]] - if unknown: - raise _TuiError( - f"Unknown family in --families: {', '.join(unknown)}. " - f"Available: {', '.join(s['catalog_by_family'])}") - chosen: Dict[str, List[dict]] = {} - family_keys: List[str] = [] - for family in requested: - if family not in family_keys: - family_keys.append(family) - chosen[family] = [opt for opt in package_dir_options( - s["catalog_by_family"][family]) if opt["recommended"]] - s["chosen"] = chosen - s["family_keys"] = family_keys - - def _compute_entries() -> None: - # Design task menu. Esc raises _GoBack, which the caller turns into - # Wizard.BACK (the design prompts are grouped: Esc returns to the - # families tree). - def task_picker(install_id: str) -> str: - result = tui.menu( - stdscr, - f"How should the '{install_id}' package be hosted?", - [ - ("design (vdes) - describe the voice with " - "--instructions", TASK_VDES), - ("tts - normal synthesis", TASK_TTS), - ], default_index=0, back_value=_GO_BACK) - if result is _GO_BACK: - raise _GoBack() - return result - - model_entries, entry_ids, install_guidance, \ - design_entry_ids, include_clone = _build_entries( - s["family_keys"], s["chosen"], s["catalog_by_family"], - task_picker, known_tasks=s["existing_tasks"]) - s.update({ - "model_entries": model_entries, - "entry_ids": entry_ids, - "install_guidance": install_guidance, - "design_entry_ids": design_entry_ids, - "include_clone": include_clone, - }) - - def _finalize() -> dict: - return { - "audiocpp_dir": s["audiocpp_dir"], - "catalog": s["catalog"], - "catalog_by_family": s["catalog_by_family"], - "output_path": s["output_path"], - "family_keys": s["family_keys"], - "chosen": s["chosen"], - "model_entries": s["model_entries"], - "entry_ids": s["entry_ids"], - "install_guidance": s["install_guidance"], - "design_entry_ids": s["design_entry_ids"], - "include_clone": s["include_clone"], - "host": s["host"], - "port": s["port"], - "backend": s["backend"], - "build": s["build"], - "lazy_load": s["lazy_load"], - "sync_port": s["sync_port"], - "sync_model_ids": s["sync_model_ids"], - "wav_dir": s["wav_dir"], - "plan": s["plan"], - "download": s["download"], - "delete_unused": s["delete_unused"], - "unused_entries": s["unused_entries"], - } - - def screen_families(): - """Pick TTS model families and packages (the modify tree).""" - tree_families = _build_tree_families(s["catalog"]) - # Modify flow: pre-check the models an existing server.json hosts, - # so the tree opens as a "modify" list rather than a fresh one. - checked_set = set() - for family, dirs in s["existing_selected"].items(): - if family not in s["catalog_by_family"]: - continue - family_index = s["catalog"].index(s["catalog_by_family"][family]) - valid_dirs = {opt["target_directory"] - for opt in package_dir_options( - s["catalog_by_family"][family])} - for target in dirs: - if target in valid_dirs: - checked_set.add((family_index, target)) - picked = tui.checkbox_tree( - stdscr, "Select TTS model families to host", - tree_families, expand_all=args.all_packages, - back_value=_GO_BACK, checked=checked_set) - if picked is _GO_BACK: - return tui.Wizard.BACK - chosen: Dict[str, List[dict]] = {} - family_keys: List[str] = [] - for family_index, option_key in picked: - family = s["catalog"][family_index]["family"] - if family not in chosen: - chosen[family] = [] - family_keys.append(family) - chosen[family].append(option_key) - for family in list(chosen): - keyed = {opt["target_directory"]: opt - for opt in package_dir_options( - s["catalog_by_family"][family])} - chosen[family] = [keyed[key] for key in chosen[family]] - s["chosen"] = chosen - s["family_keys"] = family_keys - return screen_host - - def _after_families(): - if args.families is not None: - _families_from_flag() - return screen_host - return screen_families - - def screen_host(): - """Build the model entries, then ask the bind host. - - The task/id pickers (when any) run here too and are grouped with - this screen: Esc on one of them (or on the host field) returns to - the families tree. - """ - try: - _compute_entries() - except _GoBack: - return tui.Wizard.BACK - if args.host is not None: - s["host"] = args.host - return _after_host() - host = tui.line_edit( - stdscr, "Bind host", - s["existing_host"] if isinstance(s["existing_host"], str) - else DEFAULT_HOST, - help_lines=["The IP address audiocpp will be hosted on", - "127.0.0.1 (this machine) is probably " - "correct"], back_value=_GO_BACK) - if host is _GO_BACK: - return tui.Wizard.BACK - s["host"] = host - return _after_host() - - def _after_host(): - if args.port is None: - return screen_port - s["port"] = args.port - return _after_port() - - def screen_port(): - port_text = tui.line_edit( - stdscr, "Port", - str(s["existing_port"]) if isinstance(s["existing_port"], int) - else str(config_port()), - validate=lambda s: None if (s.isdigit() - and 1 <= int(s) <= 65535) - else "Enter a port number between 1 and 65535", - help_lines=["The port audiocpp will be hosted on"], - back_value=_GO_BACK) - if port_text is _GO_BACK: - return tui.Wizard.BACK - s["port"] = int(port_text) - return _after_port() - - def _after_port(): - s["sync_port"] = None - if s["port"] != config_port(): - return screen_sync_port - return _after_sync() - - def screen_sync_port(): - sync_port = tui.confirm( - stdscr, "Update AUDIOCPP_API_URL in app/converter/config.py " - f"to port {s['port']} so audiobook.py talks to this server", - default=True, cancel_value=_GO_BACK) - if sync_port is _GO_BACK: - return tui.Wizard.BACK - s["sync_port"] = sync_port - return _after_sync() - - def _after_sync(): - if args.build_backend: - s["backend"] = args.build_backend - s["build"] = s["detected_backend"] is None - return _after_backend() - if args.backend: - s["backend"] = args.backend - s["build"] = False - return _after_backend() - if s["detected_backend"] is not None: - # Already built: use the detected backend, no menu, no build. - s["backend"] = s["detected_backend"] - s["build"] = False - return _after_backend() - # Not built for any backend yet: always ask which backend the server - # should use and offer to build it — even on a modify run, so a user - # who declined the build the first time is never stranded without a - # way to build from the TUI. - return screen_backend - - def screen_backend(): - # Pre-select the backend an existing server.json records (modify - # flow), so re-running setup lands on the previous choice. - backend_options, backend_default = _backend_options(None) - if s["existing_backend"] in BACKENDS: - backend_default = next( - (index for index, (_label, value) in enumerate(backend_options) - if value == s["existing_backend"]), backend_default) - backend = tui.menu( - stdscr, "Which inference backend should audiocpp_server " - "use?", backend_options, - default_index=backend_default, back_value=_GO_BACK) - if backend is _GO_BACK: - return tui.Wizard.BACK - s["backend"] = backend - if built_server_binary(s["audiocpp_dir"], backend) is not None: - # A checkout with builds for several backends: this one is - # already built, so there is nothing to build. - s["build"] = False - return _after_backend() - return screen_build - - def screen_build(): - # Not built for the chosen backend yet: offer to build it now. The - # build itself runs in the TUI task view (or the console tail for - # CLI runs) after the wizard. - build = tui.confirm( - stdscr, f"audiocpp_server is not built for {s['backend']}. " - f"Build it now (runs scripts/build_*)?", - default=True, cancel_value=_GO_BACK) - if build is _GO_BACK: - return tui.Wizard.BACK - s["build"] = build - return _after_backend() - - def _after_backend(): - s["lazy_load"] = True - return _after_lazy() - - def _after_lazy(): - if args.input_dir is not None: - s["wav_dir"] = args.input_dir - return _after_wav() - if s["include_clone"]: - return screen_wav - s["wav_dir"] = None - return _after_wav() - - def screen_wav(): - wav_start = detect_wav_dir(s["audiocpp_dir"], TTS_ROOT) - # Modify flow: an existing voice_dir seeds the browser so the user - # can accept it on Enter instead of re-navigating. - if isinstance(s["existing_voice_dir"], str) and s["existing_voice_dir"]: - wav_start = Path(s["existing_voice_dir"]) - wav_dir = tui.browse_directory( - stdscr, "Select the directory with your .wav voices", - info=_wav_dir_info, preview=_wav_dir_preview, - start=wav_start if wav_start is not None else VOICES_DIR, - back_value=_GO_BACK) - if wav_dir is _GO_BACK: - return tui.Wizard.BACK - s["wav_dir"] = wav_dir - return _after_wav() - - def _after_wav(): - s["plan"] = None - if s["include_clone"] and s["wav_dir"] is not None: - wav_files = find_wav_files(s["wav_dir"]) - if wav_files: - prompt_path = s["wav_dir"] / PROMPT_TEXT_FILENAME - if prompt_path.exists() and not args.force: - return screen_transcription - existing = read_prompt_text(prompt_path) if ( - prompt_path.exists() and not args.force) else {} - s["plan"] = _decide_transcription( - wav_files, existing, prompt_path.exists(), - args.force, ask_confirm) - return _after_transcription() - - def screen_transcription(): - # Transcription plan (questions only; transcription runs after). - wav_files = find_wav_files(s["wav_dir"]) - prompt_path = s["wav_dir"] / PROMPT_TEXT_FILENAME - existing = read_prompt_text(prompt_path) if ( - prompt_path.exists() and not args.force) else {} - try: - s["plan"] = _decide_transcription( - wav_files, existing, prompt_path.exists(), - args.force, ask_confirm) - except _GoBack: - return tui.Wizard.BACK - return _after_transcription() - - def _after_transcription(): - s["sync_model_ids"] = None - if len(s["entry_ids"]) == 1 and not ( - config.AUDIOCPP_MODEL_ID == s["entry_ids"][0] - and config.AUDIOCPP_CLONE_MODEL_ID == s["entry_ids"][0]): - return screen_model_sync - return _after_model_sync() - - def screen_model_sync(): - sync_model_ids = tui.confirm( - stdscr, "Update AUDIOCPP_MODEL_ID and " - "AUDIOCPP_CLONE_MODEL_ID in app/converter/config.py to " - f"'{s['entry_ids'][0]}' so audiobook.py uses this model", - default=True, cancel_value=_GO_BACK) - if sync_model_ids is _GO_BACK: - return tui.Wizard.BACK - s["sync_model_ids"] = sync_model_ids - return _after_model_sync() - - def _after_model_sync(): - new_paths = {entry["path"] for entry in s["model_entries"]} - s["unused_entries"] = unused_installed_entries( - s["output_path"], new_paths) \ - if s["existing_config"] is not None else [] - s["delete_unused"] = False - if s["unused_entries"]: - return screen_delete_unused - return _after_delete() - - def screen_delete_unused(): - delete_unused = tui.confirm( - stdscr, "Delete unused models?", default=False, - cancel_value=_GO_BACK) - if delete_unused is _GO_BACK: - return tui.Wizard.BACK - s["delete_unused"] = delete_unused - return _after_delete() - - def _after_delete(): - manager = s["audiocpp_dir"] / "tools" / "model_manager_v2.py" - if manager.is_file(): - return screen_download - s["download"] = False - return _finalize() - - def screen_download(): - # Automatic model download (or print the install commands). - try: - s["download"] = _decide_download( - s["audiocpp_dir"], s["model_entries"], ask_confirm) - except _GoBack: - return tui.Wizard.BACK - return _finalize() - - # First screen: resolve the checkout directly when it already exists - # (the modify flow), so the wizard starts on a real screen. When no - # checkout exists, clone it into ./app/audio.cpp (streaming inside the - # TUI task view, not by dropping to the console) without asking, then - # continue the same way. - audiocpp_dir = find_local_checkout() - if audiocpp_dir is None: - target = APP_DIR / AUDIOCPP_DIR_NAME - rc = taskview.run_steps(stdscr, "Clone audio.cpp", [ - taskview.TaskStep( - f"Cloning audio.cpp into {target}", - lambda emit, cancel: common.git_clone( - AUDIOCPP_GIT_URL, target, emit=emit, cancel=cancel)), - taskview.TaskStep( - "Apply ggml build patches", - lambda emit, cancel: apply_ggml_patches( - target, emit=emit, cancel=cancel)), - ]) - if rc == 130: - # Cancelled from the task view: abort the wizard quietly. - return None - if rc != 0: - raise _TuiError( - f"audio.cpp setup step failed (exit {rc}). Clone " - f"audio.cpp manually: git clone " - f"{AUDIOCPP_GIT_URL} {target}, then re-run") - audiocpp_dir = target - resolve_checkout(audiocpp_dir) - first = _after_families() - return tui.Wizard().run(first) - - -def load_server_config(server_json: Path) -> Optional[dict]: - """Read server.json into a dict, or None when it cannot be used. - - Returns None for a missing file, unreadable content, or a non-dict - document. Used by the wizard's modify flow to pre-fill its screens - from an existing config instead of prompting to overwrite it. - """ - if not server_json.exists(): - return None - try: - data = json.loads(server_json.read_text(encoding="utf-8")) - except (OSError, ValueError): - return None - if not isinstance(data, dict): - return None - return data - - -def server_config_selections(server_config: dict, - catalog: List[dict] - ) -> Tuple[Dict[str, List[str]], - Dict[Tuple[str, str], str]]: - """Map an existing server.json's models back to catalog selections. - - Returns ``(selected_dirs, tasks)``: ``selected_dirs`` maps a catalog - family to the target directories it hosts (``models/<target>`` paths - with the ``models/`` prefix stripped, in server.json order), and - ``tasks`` maps ``(family, target_directory)`` to the entry's task - (``"tts"`` or ``"vdes"``) so the wizard can preserve how design - packages were hosted. Entries whose family is not in the CATALOG are - ignored — the wizard cannot offer them again. - """ - families = {entry["family"] for entry in catalog} - selected_dirs: Dict[str, List[str]] = {} - tasks: Dict[Tuple[str, str], str] = {} - for entry in server_config.get("models") or []: - if not isinstance(entry, dict): - continue - family = entry.get("family") - if not isinstance(family, str) or family not in families: - continue - path = entry.get("path") - if not isinstance(path, str): - continue - target = path[len("models/"):] if path.startswith("models/") else path - if family not in selected_dirs: - selected_dirs[family] = [] - if target not in selected_dirs[family]: - selected_dirs[family].append(target) - tasks[(family, target)] = str(entry.get("task") or TASK_TTS) - return selected_dirs, tasks - - -def _model_path_present(path: Path) -> bool: - """True when a server.json model path holds actual model files. - - A present path is either a file (a single-model package) or a non-empty - directory (the usual GGUF package target directory; an empty one means a - download that never ran or was cleaned up halfway). - """ - try: - if path.is_file(): - return True - if path.is_dir(): - return any(path.iterdir()) - except OSError: - return False - return False - - -def _all_models_present(audiocpp_dir: Path, model_entries: List[dict]) -> bool: - """True when every selected model entry's path already holds files on disk. - - Paths resolve against the checkout (where model_manager_v2.py installs - them), honoring absolute paths. Used by the wizard to skip the - "Automatically download the selected models" prompt when nothing is - actually missing. An empty selection is treated as not-present. - """ - if not model_entries: - return False - for entry in model_entries: - rel = entry.get("path") - if not isinstance(rel, str) or not rel: - return False - path = Path(rel) if Path(rel).is_absolute() else audiocpp_dir / rel - if not _model_path_present(path): - return False - return True - - -def missing_model_entries(server_json: Path) -> List[dict]: - """Return the server.json model entries whose files are not on disk. - - Paths resolve exactly like audiocpp_server resolves them (relative paths - against the server.json's directory). Each returned entry carries the - entry ``id`` and ``rel`` (the configured path string); used by ``detect`` - to warn that a conversion would fail until the models are installed. - """ - try: - data = json.loads(server_json.read_text(encoding="utf-8")) - except (OSError, ValueError): - return [] - if not isinstance(data, dict): - return [] - base = server_json.parent - missing: List[dict] = [] - for entry in data.get("models") or []: - if not isinstance(entry, dict): - continue - rel = entry.get("path") - if not isinstance(rel, str) or not rel: - continue - path = Path(rel) if Path(rel).is_absolute() else base / rel - if _model_path_present(path): - continue - missing.append({"id": str(entry.get("id") or rel), "rel": rel}) - return missing - - -def _install_id_by_path(audiocpp_dir: Path) -> Dict[str, str]: - """Map ``models/<target_directory>`` -> catalog install id. - - The catalog package that installs a model is derived from the - ``default_path`` of each TTS family; an entry whose path matches no - catalog package has no install id. - """ - by_path: Dict[str, str] = {} - try: - for entry in load_model_catalog(audiocpp_dir): - by_path[entry["default_path"]] = entry["install_id"] - except (NotADirectoryError, OSError): - pass - return by_path - - -def installed_model_entries(server_json: Path) -> List[dict]: - """Return the server.json model entries whose files ARE on disk. - - The complement of ``missing_model_entries``: each returned entry carries - the entry ``id`` and ``rel`` (the configured path string), resolved - exactly like ``missing_model_entries`` (relative against the server.json's - directory). Used by the wizard's "Delete unused models?" step to find - already-downloaded models that were unselected. - """ - try: - data = json.loads(server_json.read_text(encoding="utf-8")) - except (OSError, ValueError): - return [] - if not isinstance(data, dict): - return [] - base = server_json.parent - installed: List[dict] = [] - for entry in data.get("models") or []: - if not isinstance(entry, dict): - continue - rel = entry.get("path") - if not isinstance(rel, str) or not rel: - continue - path = Path(rel) if Path(rel).is_absolute() else base / rel - if _model_path_present(path): - installed.append({"id": str(entry.get("id") or rel), "rel": rel}) - return installed - - -def missing_model_install_guidance(audiocpp_dir: Path, - missing: List[dict]) -> List[Tuple[str, str]]: - """Map MISSING model entries to (display name, install id) pairs. - - The install id is derived from each entry's configured path via the - catalog (see ``_install_id_by_path``); entries whose path matches no - catalog package are skipped (there is no ``model_manager_v2.py install`` - command for them). Feeds ``_install_models`` for the "Download Missing - Models" action. - """ - by_path = _install_id_by_path(audiocpp_dir) - guidance: List[Tuple[str, str]] = [] - for item in missing: - install_id = by_path.get(item["rel"]) - if install_id: - guidance.append((item["id"], install_id)) - return guidance - - -def model_install_hints(audiocpp_dir: Path, - missing: List[dict]) -> List[str]: - """Remediation lines for MISSING model entries (see missing_model_entries). - - Maps each entry's configured path back to the catalog package that - installs it (``models/<target_directory>`` -> install id) so the line - carries the exact ``model_manager_v2.py install`` command; entries whose - directory matches no catalog package just name the path. - """ - by_path = _install_id_by_path(audiocpp_dir) - hints: List[str] = [] - for item in missing: - install_id = by_path.get(item["rel"]) - hint = f"model not downloaded: {item['id']} ({item['rel']})" - if install_id: - hint += (f" — install with: python tools/model_manager_v2.py " - f"install {install_id}") - hints.append(hint) - return hints - - -def install_models(audiocpp_dir: Path, - guidance: List[Tuple[str, str]], - emit=None, cancel=None) -> int: - """Download the (display name, install id) models via the helper script. - - Runs ``model_manager_v2.py install`` for each de-duped install id in the - checkout, streaming to the console (or to EMIT, the in-TUI task view); a - failing install is reported as a warning and does not abort the rest. - Returns 0 when every download succeeded, 130 when cancelled, 1 when any - failed. Used by the hub's "Download Missing Models" action (see - ``missing_model_install_guidance`` for the mapping). - """ - return _install_models(audiocpp_dir, guidance, download=True, - emit=emit, cancel=cancel) - - -def hand_install_guidance(audiocpp_dir: Path, - missing: List[dict]) -> str: - """Explain how to install MISSING model entries by hand. - - Returns a multi-line message listing each missing model's id and the - path its files must be placed in (``rel``, resolved against the - checkout). Used when the missing models cannot be mapped to - a ``model_manager_v2.py install`` command, so the user still knows what - to download and where to put it. - """ - lines = [ - "None of the missing models map to a model_manager_v2.py install " - "command.", - "Download them by hand and place the files at these paths:", - ] - for item in missing: - lines.append(f" {item['id']} -> {item['rel']}") - lines.append(f"(paths are relative to {audiocpp_dir})") - return "\n".join(lines) - - -def unused_installed_entries(server_json: Path, - new_paths: Set[str]) -> List[dict]: - """Return installed server.json entries whose path is not in NEW_PATHS. - - The already-downloaded models (see ``installed_model_entries``) that the - new selection does not host any more — the candidates for the wizard's - "Delete unused models?" prompt. Entries whose files are not on disk are - never listed (there is nothing to delete). - """ - return [entry for entry in installed_model_entries(server_json) - if entry["rel"] not in new_paths] - - -def delete_model_files(server_json: Path, entries: List[dict]) -> int: - """Remove the on-disk model files for ENTRIES ({id, rel}) from disk. - - Each entry's ``rel`` is resolved exactly like the server resolves it - (relative against ``server_json``'s directory; absolute paths honored), - then removed as a directory tree or a single file. Missing entries are - ignored. Returns the number of paths removed. Used by the wizard's - "Delete unused models?" step — the regenerated server.json already only - lists the kept models, so no entry cleanup is needed here. - """ - base = server_json.parent - removed = 0 - for item in entries: - rel = item.get("rel") - if not isinstance(rel, str) or not rel: - continue - path = Path(rel) if Path(rel).is_absolute() else base / rel - try: - if not path.exists(): - continue - if path.is_dir(): - shutil.rmtree(path, ignore_errors=True) - else: - path.unlink() - except OSError as exc: - print(f"[WARNING] Could not remove {path}: {exc}") - continue - print(f"[OK] Removed unused model {path}") - removed += 1 - return removed - - -def uninstall(*, emit=None, cancel=None) -> int: - """Remove the audio.cpp backend entirely: stop its server, delete the checkout. - - The checkout (``app/audio.cpp``) holds the built binary, the downloaded - models, and the server.json, so removing the directory uninstalls the - backend. A running server this tool started is stopped first - (best-effort). - - EMIT is accepted for registry symmetry with the other backends but is - unused here — this uninstall has no subprocess phase, and its prints are - captured by the task view when run in the TUI. CANCEL is a - ``threading.Event`` honored between phases only (after the server has - been stopped, before the checkout is deleted), so a started phase always - completes and the uninstall never tears halfway. Returns the exit code - (130 when cancelled before a remaining phase). - """ - # Only stop when a pid file exists: without one this tool never - # started the server, so the "not started by this tool" notice would - # be uninstall-time noise. - if servers.pid_for("audiocpp") is not None: - servers.stop("audiocpp") - if common.cancel_requested(cancel): - return 130 - checkout = find_local_checkout() - if checkout is None: - print("[INFO] No audio.cpp checkout to remove.") - return 0 - print(f"[INFO] Removing audio.cpp checkout {checkout}...") - shutil.rmtree(checkout, ignore_errors=True) - print("[OK] audio.cpp removed.") - return 0 - - -def find_local_checkout() -> Optional[Path]: - """Return the managed audio.cpp checkout at ``app/audio.cpp``. - - Returns the path only when it contains a ``model_specs`` directory; - the checkout is installed there by the setup wizard and nowhere else. - """ - try: - resolved = (APP_DIR / AUDIOCPP_DIR_NAME).resolve() - except OSError: - return None - if (resolved / "model_specs").is_dir(): - return resolved - return None - - -def fetch_server_models(api_url: str) -> Optional[List[Dict[str, str]]]: - """List a running audiocpp_server's model entries via GET /v1/models. - - Returns ``[{id, family, task}, ...]`` — the same shape the converter's - client resolves at startup — or None when URL does not answer with a - valid document (wrong server, still starting, older audio.cpp). Used by - the hub to drive the convert menus against a remote server that has no - local server.json describing it. - """ - try: - with urllib.request.urlopen( - f"{api_url.rstrip('/')}/v1/models", timeout=10) as response: - payload = json.loads(response.read().decode("utf-8")) - except (OSError, ValueError): - # URLError/HTTPError/socket errors are OSErrors; a non-JSON body is - # a ValueError. Anything else means "not an audiocpp_server". - return None - entries = payload.get("data") if isinstance(payload, dict) else None - models: List[Dict[str, str]] = [] - for entry in entries or []: - if isinstance(entry, dict) and entry.get("id"): - models.append({ - "id": str(entry["id"]), - "family": str(entry.get("family") or ""), - "task": str(entry.get("task") or ""), - }) - return models - - -def fetch_server_voices(api_url: str, model_id: str) -> Optional[List[str]]: - """List a running audiocpp_server's voices for MODEL_ID. - - Queries ``GET /v1/audio/voices?model=<id>`` — the endpoint the converter - validates ``--voice`` against — and returns its voice-name list, or None - when the server cannot be queried. Lets the hub offer a remote server's - voices without reading its configuration locally. - """ - query = urllib.parse.urlencode({"model": model_id}) - try: - with urllib.request.urlopen( - f"{api_url.rstrip('/')}/v1/audio/voices?{query}", - timeout=10) as response: - payload = json.loads(response.read().decode("utf-8")) - except (OSError, ValueError): - return None - voices = payload.get("voices") if isinstance(payload, dict) else None - if not isinstance(voices, list): - return None - return [str(voice) for voice in voices] - - -def find_audiocpp_server_bin(audiocpp_dir: Path) -> Optional[Path]: - """Return the built audiocpp_server binary, or None when not built. - - Scans ``audiocpp_dir/build/*`` for a build directory containing - ``bin/audiocpp_server`` (``.exe`` allowed on Windows). When several - builds exist the first (alphabetical) is returned. - """ - build_root = audiocpp_dir / "build" - if not build_root.is_dir(): - return None - try: - build_dirs = sorted(build_root.iterdir(), - key=lambda p: p.name.lower()) - except OSError: - return None - for build_dir in build_dirs: - if not build_dir.is_dir(): - continue - for name in ("audiocpp_server", "audiocpp_server.exe"): - server = build_dir / "bin" / name - if server.exists(): - return server - return None - - -def built_server_binary(audiocpp_dir: Path, backend: str) -> Optional[Path]: - """Return the built audiocpp_server for BACKEND, or None. - - Like ``find_audiocpp_server_bin`` but limited to build directories whose - name carries the BACKEND token (``-cuda-``, ``-vulkan-``, ``-hip-``, - ``-cpu-``; ``-metal-`` counts as ``cpu``). A checkout with builds for - several backends is asked which one to use without re-offering a build - for a backend that is already built. - """ - build_root = audiocpp_dir / "build" - if not build_root.is_dir(): - return None - try: - build_dirs = sorted(build_root.iterdir(), - key=lambda p: p.name.lower()) - except OSError: - return None - for build_dir in build_dirs: - if not build_dir.is_dir(): - continue - match = _BACKEND_TOKEN_RE.search(build_dir.name.lower()) - if not match: - continue - token = "cpu" if match.group(1) == "metal" else match.group(1) - if token != backend: - continue - for name in ("audiocpp_server", "audiocpp_server.exe"): - server = build_dir / "bin" / name - if server.exists(): - return server - return None - - -def find_build_script(audiocpp_dir: Path) -> Optional[Path]: - """Return the audio.cpp build helper script to run, or None. - - Prefers ``scripts/build_linux.sh``; otherwise the first - ``scripts/build_*.sh`` it finds. (Windows ``.bat`` scripts are not run - automatically — build manually there.) - """ - scripts = audiocpp_dir / "scripts" - if not scripts.is_dir(): - return None - preferred = scripts / "build_linux.sh" - if preferred.exists(): - return preferred - try: - candidates = sorted(scripts.glob("build_*.sh"), - key=lambda p: p.name.lower()) - except OSError: - return None - return candidates[0] if candidates else None - - -# Each entry pairs a shipped patch with the vendored file it touches and a -# regex marker proving the fix is already present (so the patch is skipped -# idempotently once applied, or once the fork re-vendors a fixed ggml). -GGML_PATCHES = [ - { - "file": "ggml-top-k-cuda-iterator.patch", - "target": "external/ggml/src/ggml-cuda/top-k.cu", - "marker": r"#\s*include\s*<cuda/iterator>", - "label": "top-k.cu: add #include <cuda/iterator> (CCCL 3.x build fix)", - }, -] - - -def apply_ggml_patches(audiocpp_dir: Path, *, emit=None, cancel=None) -> int: - """Apply the shipped ggml build patches to an audio.cpp checkout. - - Idempotent: a patch whose marker already matches its target is skipped - (it is either already applied, or the fork re-vendored a fixed ggml). A - patch that no longer applies because the vendored file changed shape is a - loud, non-interactive failure — the build is aborted so the user - re-evaluates the patch instead of hitting a known nvcc break minutes - later. Returns 0 when every patch is applied or already present, 1 on - drift, 130 when cancelled. - """ - for patch in GGML_PATCHES: - if cancel is not None and cancel.is_set(): - return 130 - target = audiocpp_dir / patch["target"] - if not target.is_file(): - print(f"[INFO] {patch['file']}: target {patch['target']} not " - f"present in this checkout; skipping") - continue - try: - text = target.read_text(encoding="utf-8", errors="ignore") - except OSError as exc: - print(f"[WARNING] {patch['file']}: could not read {target}: " - f"{exc}; skipping") - continue - if re.search(patch["marker"], text): - print(f"[OK] {patch['file']}: fix already present, skipping") - continue - patch_path = PATCH_DIR / patch["file"] - if not patch_path.is_file(): - print(f"[ERROR] {patch['file']}: patch file not found at " - f"{patch_path}; cannot apply") - return 1 - check_argv = ["git", "-C", str(audiocpp_dir), "apply", "--check", - "--whitespace=nowarn", str(patch_path)] - check_rc = common.run_console_subprocess( - check_argv, emit=emit, cancel=cancel) - if check_rc == 130 or (cancel is not None and cancel.is_set()): - return 130 - if check_rc != 0: - print(f"[ERROR] {patch['file']}: no longer applies to " - f"{patch['target']} (git apply --check exit {check_rc}). " - f"The audio.cpp fork's vendored ggml changed shape and " - f"still lacks the fix. Re-evaluate {patch_path}: " - f"regenerate the patch, or drop this entry if the fork " - f"now ships the fix.") - return 1 - apply_argv = ["git", "-C", str(audiocpp_dir), "apply", - "--whitespace=nowarn", str(patch_path)] - rc = common.run_console_subprocess( - apply_argv, emit=emit, cancel=cancel) - if rc == 130 or (cancel is not None and cancel.is_set()): - return 130 - if rc != 0: - print(f"[ERROR] {patch['file']}: git apply failed (exit {rc})") - return rc - print(f"[OK] {patch['file']}: applied ({patch['label']})") - return 0 - - -def build_audiocpp(audiocpp_dir: Path, backend: str, *, - emit=None, cancel=None) -> int: - """Build audiocpp_server for BACKEND, streaming output. - - With EMIT None the build script runs on the console (inherits the - terminal); with EMIT given (the in-TUI task view) its output streams line - by line to EMIT so the view can show progress, and CANCEL aborts it. - - On the EMIT (TUI) path the build output is also tee'd to - ``app/logs/audiocpp_build_<timestamp>.log`` so it survives the curses - session; when the build fails (and was not cancelled) a post-TUI notice - with the copy-pastable command and the log path is queued for the console - (see ``backends.common.record_post_tui_notice``). - - Returns the build script's exit code (non-zero when the script is - missing). - """ - script = find_build_script(audiocpp_dir) - if script is None: - message = (f"[ERROR] No build script found in {audiocpp_dir}/scripts; " - "build audiocpp_server manually (see the audio.cpp README)") - print(message) - if emit is not None: - common.record_post_tui_notice(message) - return 1 - argv = ["sh", str(script), "--backend", backend, "--target", - "audiocpp_server", "--deployment-build"] - command = f"cd {audiocpp_dir} && {shlex.join(argv)}" - if emit is None: - print(f"[INFO] Building audiocpp_server for {backend} ({command})...") - patch_rc = apply_ggml_patches(audiocpp_dir, cancel=cancel) - if patch_rc == 130 or (cancel is not None and cancel.is_set()): - return 130 - if patch_rc != 0: - print("[ERROR] ggml build patches could not be applied; " - "aborting audiocpp_server build. See the messages above " - "and re-evaluate app/backends/patches/.") - return patch_rc - return common.run_console_subprocess(argv, cwd=audiocpp_dir) - return _build_audiocpp_tui(emit, cancel, argv, command, audiocpp_dir) - - -def _build_audiocpp_tui(emit, cancel, argv: List[str], command: str, - audiocpp_dir: Path) -> int: - """Run the build on the TUI path: tee output to a log file. - - The ggml patch step runs first, inside the same log: every emitted - line (patch status, build output) is also written (and flushed) to - ``app/logs/audiocpp_build_<timestamp>.log``. On failure a summary (the - copy-pastable COMMAND and the log path) is emitted into the TUI, - written to the log, and queued as a post-TUI console notice. A - cancelled build (CANCEL set) is not reported as a failure, but its - partial output stays in the log file. - """ - log_path = common.LOG_DIR / ( - f"audiocpp_build_{datetime.now():%Y%m%d_%H%M%S}.log") - log_path.parent.mkdir(parents=True, exist_ok=True) - log_handle = log_path.open("w", encoding="utf-8") - - def tee(line: str) -> None: - log_handle.write(line + "\n") - log_handle.flush() - emit(line) - - class _TeeWriter(io.TextIOBase): - """Route print() output from the patch step into the log too.""" - - def write(self, s: str) -> int: - for line in s.splitlines(): - if line: - tee(line) - return len(s) - - try: - with contextlib.redirect_stdout(_TeeWriter()): - patch_rc = apply_ggml_patches(audiocpp_dir, emit=tee, - cancel=cancel) - if patch_rc == 130 or (cancel is not None and cancel.is_set()): - return 130 - if patch_rc != 0: - notice = ("[ERROR] ggml build patches could not be applied; " - "aborting audiocpp_server build. See the messages " - "above and re-evaluate app/backends/patches/.") - tee(notice) - common.record_post_tui_notice(notice) - return patch_rc - tee(f"[INFO] Building audiocpp_server ({command})...") - rc = common.run_console_subprocess( - argv, cwd=audiocpp_dir, emit=tee, cancel=cancel) - if rc != 0 and (cancel is None or not cancel.is_set()): - notice = (f"[ERROR] audio.cpp build failed (exit code {rc}).\n" - f" Build log: {log_path}\n" - f" Troubleshoot by re-running this command:\n" - f" {command}") - for line in notice.splitlines(): - tee(line) - common.record_post_tui_notice(notice) - finally: - log_handle.close() - return rc - - -def _print_launch_hint(audiocpp_dir: Path, output_path: Path) -> None: - """Print remediation when audiocpp_server is missing (troubleshooting). - - The hub starts and stops the server itself, so a working install gets - no manual launch instructions. When no binary was built, though, the - user needs to know how to build and run it by hand. The commands are - prefixed with ``cd <checkout> &&`` because the server discovers - model_specs/<family>.json relative to its working directory. - """ - if find_audiocpp_server_bin(audiocpp_dir) is not None: - return - print("\n[INFO] audiocpp_server binary not found. Build it first, e.g.:") - script = find_build_script(audiocpp_dir) - if script is not None: - print(f" sh {script} --backend <cuda|vulkan|hip|cpu> " - "--target audiocpp_server --deployment-build") - print(f" then run: cd {audiocpp_dir} && ./build/<platform>-<backend>" - f"-release/bin/audiocpp_server --config {output_path}") - - -def _execute_lanes(settings: dict, - args: argparse.Namespace) -> List[taskview.TaskLane]: - """Build the ordered setup steps for the in-TUI task view, per lane. - - The same work ``_execute`` runs on the console, split into two lanes so - the view can run the build in one pane while configuring and downloading - models in the other (both progress bars visible at once). The build lane - exists only when ``settings["build"]`` is set; the models lane always - exists (transcribe → write server.json → download/print commands). - Shared results (the transcription mapping) travel through a small closure - dict scoped to the models lane. Each step's ``work(emit, cancel)`` - returns its exit code; subprocess steps stream through EMIT and abort on - CANCEL, while print()-based steps are captured by the view's stdout - routing. - """ - audiocpp_dir = settings["audiocpp_dir"] - state: dict = {} - build = settings.get("build") - lanes: List[taskview.TaskLane] = [] - - if build: - def build_step(emit, cancel): - rc = build_audiocpp(audiocpp_dir, settings["backend"], - emit=emit, cancel=cancel) - if rc != 0: - print(f"[WARNING] build exited with code {rc}; the server.json " - "was still written — build audiocpp_server manually " - "before starting it") - else: - print("[OK] build complete") - return rc - lanes.append(taskview.TaskLane( - "Build", - [taskview.TaskStep( - f"Build audiocpp_server ({settings['backend']})", - build_step)])) - - def transcribe(emit, cancel): - args.input_dir = settings["wav_dir"] - if settings["include_clone"] and args.input_dir is not None: - transcripts, write_prompt = _transcribe( - args, plan=settings["plan"], cancel=cancel) - elif args.input_dir is not None: - print(f"[WARNING] Ignoring {args.input_dir}: no clone-capable " - "family selected, so voice presets are not used") - transcripts, write_prompt = {}, False - else: - transcripts, write_prompt = {}, False - state["transcripts"] = transcripts - state["write_prompt"] = write_prompt - return 0 - - def write(emit, cancel): - # Port sync (applied now that the terminal is back). - if settings["sync_port"] is True: - _apply_port_sync(settings["port"], True) - elif settings["sync_port"] is False: - _apply_port_sync(settings["port"], False) - - _write_and_advise( - audiocpp_dir, settings["wav_dir"], settings["output_path"], - settings["model_entries"], settings["install_guidance"], - settings["host"], settings["port"], settings["backend"], - settings["lazy_load"], state["transcripts"], state["write_prompt"]) - - # Delete-unused cleanup (modify flow): remove the already-downloaded - # models the new selection dropped. The regenerated server.json - # already only lists the kept entries. - if settings.get("delete_unused"): - removed = delete_model_files(settings["output_path"], - settings["unused_entries"]) - print(f"[OK] Deleted {removed} unused model " - f"{'entry' if removed == 1 else 'entries'} from disk.") - - if len(settings["entry_ids"]) == 1: - _offer_config_model_id_sync(settings["entry_ids"][0], - settings["sync_model_ids"]) - print_empty_transcript_warning(state["transcripts"]) - return 0 - - def install(emit, cancel): - _install_models(audiocpp_dir, settings["install_guidance"], - settings["download"], emit=emit, cancel=cancel) - _print_launch_hint(audiocpp_dir, settings["output_path"]) - return 0 - install_title = "Download models" if settings.get("download") \ - else "Print model install commands" - - lanes.append(taskview.TaskLane( - "Configure & download", - [taskview.TaskStep("Transcribe reference voices", transcribe), - taskview.TaskStep("Write server.json & sync config", write), - taskview.TaskStep(install_title, install)])) - - return lanes - - -def _execute_steps(settings: dict, - args: argparse.Namespace) -> List[taskview.TaskStep]: - """The ordered setup steps for the sequential console path. - - The lanes ``_execute_lanes`` builds, flattened into one ordered list - (build first, then transcribe → write → download), so the console tail - is byte-identical to the pre-lanes behavior. - """ - steps: List[taskview.TaskStep] = [] - for lane in _execute_lanes(settings, args): - steps.extend(lane.steps) - return steps - - -def _execute(settings: dict, args: argparse.Namespace) -> int: - """Shared console tail: build, sync, transcribe, write, install, advise. - - Runs after the TUI wizard returns (or after _collect_from_flags for a - non-interactive run): the terminal is plain, so subprocess output and - transcription progress appear normally. The same work as - ``_execute_steps``, run with no emit (console streaming). - """ - return taskview.run_steps_inline(_execute_steps(settings, args)) - - -def setup_screen(stdscr) -> int: - """Run the setup wizard on an existing curses screen (the hub's). - - The hub drives this as one screen of its own ``tui.Wizard`` stack, so - Esc on the wizard's first screen simply returns here and the hub pops - back to the menu that launched it. The setup tail (build, transcribe, - write, download) runs inside the TUI task view on this same screen, so - the hub's curses session stays intact and the user sees per-step status - and progress instead of being dropped to the console. On a fresh install - the build and the model setup run as two parallel lanes (a split view), - so cloning → configuring → building+downloading is one continuous, - one-click flow; the individual "Build" and "Download Missing Models" hub - actions remain only as fallbacks when something fails or is interrupted. - Returns 0 on completion, 1 when the user aborted. - """ - parser = build_parser() - args = parser.parse_args([]) - settings = _wizard(stdscr, args, parser) - if settings is None: - return 1 - return taskview.run_lanes(stdscr, "Setting up audio.cpp", - _execute_lanes(settings, args)) - - -def build_screen(stdscr) -> int: - """Build audiocpp_server from the hub when the checkout has no binary. - - Asks which backend to build for (pre-selecting the backend an existing - server.json records, else cuda), runs the build inside the TUI task view - — alongside a download of any missing models when server.json is already - configured and those models map to an install command (the split view), - or just the build otherwise — then updates server.json's ``backend`` - field to match. Returns 0 on success, non-zero when the user backed out, - cancelled, or the build failed. This is the hub's "Build audio.cpp - server" action, so a checkout that was cloned but never built is always - buildable from the TUI; the standalone "Download Missing Models" action - stays as the fallback when the download fails or is interrupted. - """ - checkout = find_local_checkout() - if checkout is None: - tui.flash(stdscr, "No audio.cpp checkout found — install audio.cpp " - "first.", "err") - return 1 - if find_audiocpp_server_bin(checkout) is not None: - tui.flash(stdscr, "audiocpp_server is already built.", "ok") - return 0 - server_config = load_server_config(checkout / "server.json") or {} - recorded = server_config.get("backend") - options, default = _backend_options(None) - if recorded in BACKENDS: - default = next((i for i, (_label, value) in enumerate(options) - if value == recorded), default) - backend = tui.menu( - stdscr, "Which inference backend should audiocpp_server be built " - "for?", options, default_index=default, back_value=_GO_BACK) - if backend is _GO_BACK: - return 1 - - def build_step(emit, cancel): - return build_audiocpp(checkout, backend, emit=emit, cancel=cancel) - - lanes = [taskview.TaskLane( - "Build", [taskview.TaskStep( - f"Build audiocpp_server ({backend})", build_step)])] - - # Missing models this build can also fetch, so a configured backend that - # lost its binary is restored to "installed" in one step. - server_json = checkout / "server.json" - missing = missing_model_entries(server_json) if server_json.exists() else [] - guidance = missing_model_install_guidance(checkout, missing) \ - if missing else [] - - if guidance: - def download_step(emit, cancel): - install_models(checkout, guidance, emit=emit, cancel=cancel) - return 0 - lanes.append(taskview.TaskLane( - "Download models", - [taskview.TaskStep("Download missing models", download_step)])) - - title = "Build & download models" if len(lanes) == 2 \ - else "Build audiocpp_server" - rc = taskview.run_lanes(stdscr, title, lanes) - if rc != 0: - return rc - if update_server_backend(backend): - tui.flash(stdscr, f"audiocpp_server built for {backend}.", "ok") - else: - tui.flash(stdscr, f"audiocpp_server built for {backend}. (Could not " - "update server.json's backend field — reconfigure audio.cpp " - "if it was already configured.)", "warn") - # Models that can't be mapped to an install command still need hand - # installation; say so now rather than leaving the user in the dark. - if missing and not guidance: - tui.flash(stdscr, hand_install_guidance(checkout, missing), "err") - return 0 - - -def update_server_backend(backend: str) -> bool: - """Rewrite the 'backend' in the checkout's server.json, or True when none. - - Sets ``backend`` to BACKEND in ``<checkout>/server.json`` (same - ``json.dump`` formatting as the wizard). Returns True when the file now - carries BACKEND, when there is no server.json (nothing to sync), or when - it already does; False when the file exists but cannot be read/written. - """ - checkout = find_local_checkout() - if checkout is None: - return True - server_json = checkout / "server.json" - if not server_json.exists(): - return True - try: - data = json.loads(server_json.read_text(encoding="utf-8")) - except (OSError, ValueError): - return False - if not isinstance(data, dict): - return False - if data.get("backend") == backend: - return True - data["backend"] = backend - try: - with server_json.open("w", encoding="utf-8") as handle: - json.dump(data, handle, indent=2, ensure_ascii=False) - handle.write("\n") - except OSError: - return False - return True - - -def run_tui(args: Optional[argparse.Namespace] = None, - parser: Optional[argparse.ArgumentParser] = None) -> int: - """Run the audio.cpp setup wizard end-to-end. - - With no ARGS (the hub's call) a default namespace is built so the full - wizard runs. Called from ``main`` after argparse when the terminal is - interactive. Returns the process exit code. - """ - import curses - if args is None: - parser = build_parser() - args = parser.parse_args([]) - if args.input_dir is not None and not args.input_dir.is_dir(): - print(f"[ERROR] --wavs not found: {args.input_dir}", - file=sys.stderr) - return 2 - try: - settings = curses.wrapper(_wizard, args, parser) - except _TuiError as exc: - print(f"[ERROR] {exc}", file=sys.stderr) - return 2 - except tui.WizardCancelled: - print("\n[INFO] Cancelled; nothing was written") - return 1 - try: - curses.curs_set(1) # restore the text cursor hidden by the TUI - except curses.error: - pass - if settings is None: - print("[INFO] Aborted; existing server.json kept") - return 1 - return _execute(settings, args) - - -def _collect_from_flags(args: argparse.Namespace, - parser: argparse.ArgumentParser) -> Optional[dict]: - """Build the settings dict from flags for a non-interactive run. - - Every required value must come from a flag (there are no prompts in a - non-interactive run); a missing one is a hard ``parser.error``. Returns - the settings dict, or None when the user declined an overwrite (the - default-location fallback then also exists). - """ - # Checkout: ./app/audio.cpp, else --clone clones one there. - audiocpp_dir = find_local_checkout() - if audiocpp_dir is None and args.clone: - target = APP_DIR / AUDIOCPP_DIR_NAME - rc = common.git_clone(AUDIOCPP_GIT_URL, target) - if rc != 0: - parser.error(f"git clone failed (exit {rc}); clone audio.cpp " - f"manually: git clone {AUDIOCPP_GIT_URL} {target}") - patch_rc = apply_ggml_patches(target) - if patch_rc != 0: - parser.error( - f"ggml build patches could not be applied to {target} " - f"(exit {patch_rc}); see messages above. The audio.cpp " - f"fork's vendored ggml may have changed — re-evaluate " - f"app/backends/patches/.") - audiocpp_dir = target - if audiocpp_dir is None: - parser.error( - "An audio.cpp checkout is required. Pass --clone to clone " - "app/audio.cpp, or run without flags for the TUI wizard.") - try: - catalog = load_model_catalog(audiocpp_dir) - except NotADirectoryError as exc: - parser.error(str(exc)) - if not catalog: - parser.error( - f"No TTS model families found in {audiocpp_dir}/model_specs; " - "check the checkout is up to date") - catalog_by_family = {entry["family"]: entry for entry in catalog} - - # Families: required from --families in a non-interactive run. - if args.families is None: - parser.error("--families is required in a non-interactive run (or run " - "without flags for the TUI wizard)") - requested = [f.strip() for f in args.families.split(",") if f.strip()] - unknown = [f for f in requested if f not in catalog_by_family] - if unknown: - parser.error( - f"Unknown family in --families: {', '.join(unknown)}. " - f"Available: {', '.join(catalog_by_family)}") - family_keys: List[str] = [] - for fam in requested: - if fam not in family_keys: - family_keys.append(fam) - - chosen: Dict[str, List[dict]] = {} - for family in family_keys: - opts = package_dir_options(catalog_by_family[family]) - if args.all_packages: - chosen[family] = opts - else: - chosen[family] = [opt for opt in opts if opt["recommended"]] - - # Non-interactive picker: design packages default to vdes. - def task_picker(install_id: str) -> str: - return TASK_VDES - - model_entries, entry_ids, install_guidance, design_entry_ids, include_clone = \ - _build_entries(family_keys, chosen, catalog_by_family, - task_picker) - - # Server settings. - host = args.host or DEFAULT_HOST - detected_backend = detect_backend(audiocpp_dir) - if args.build_backend: - backend = args.build_backend - build = detected_backend is None - elif args.backend: - backend = args.backend - build = False - elif detected_backend is not None: - backend = detected_backend - build = False - else: - backend = "cuda" - build = False - port = args.port if args.port is not None else config_port() - lazy_load = True - - # Output path / overwrite (decline falls back to cwd, then aborts). - output_path = args.output if args.output is not None \ - else audiocpp_dir / "server.json" - if output_path.exists() and not args.force: - if args.output is None: - output_path = Path.cwd() / "server.json" - if output_path.exists() and not args.force: - print("[INFO] Aborted; existing server.json kept") - return None - else: - print("[INFO] Aborted; existing server.json kept") - return None - - # Config sync decisions (auto-apply unless explicitly declined). - sync_port: Optional[bool] = None - if port != config_port(): - sync_port = not args.no_sync_port - sync_model_ids: Optional[bool] = None - if len(entry_ids) == 1 and not ( - config.AUDIOCPP_MODEL_ID == entry_ids[0] - and config.AUDIOCPP_CLONE_MODEL_ID == entry_ids[0]): - sync_model_ids = not args.no_sync_model_ids - - # Wav dir + transcription plan (defaults to the project's voices/ dir). - wav_dir = args.input_dir if args.input_dir is not None else VOICES_DIR - plan: Optional[dict] = None - if include_clone and wav_dir is not None: - wav_files = find_wav_files(wav_dir) - if wav_files: - prompt_path = wav_dir / PROMPT_TEXT_FILENAME - plan = _flag_plan(wav_files, prompt_path, args.force) - - return { - "audiocpp_dir": audiocpp_dir, - "catalog": catalog, - "catalog_by_family": catalog_by_family, - "output_path": output_path, - "family_keys": family_keys, - "chosen": chosen, - "model_entries": model_entries, - "entry_ids": entry_ids, - "install_guidance": install_guidance, - "design_entry_ids": design_entry_ids, - "include_clone": include_clone, - "host": host, - "port": port, - "backend": backend, - "build": build, - "lazy_load": lazy_load, - "sync_port": sync_port, - "sync_model_ids": sync_model_ids, - "wav_dir": wav_dir, - "plan": plan, - "download": args.download, - } - - -def build_parser() -> argparse.ArgumentParser: - """The audio.cpp setup CLI (also used to build a default namespace).""" - parser = argparse.ArgumentParser( - description="Set up the audio.cpp TTS backend: clone/build, pick " - "models, write server.json, and sync app/converter/config.py.") - parser.add_argument("--wavs", type=resolve_wav_dir_arg, default=None, - dest="input_dir", metavar="WAV_DIR", - help="Directory with .wav reference files to publish as " - "a server-level voice_dir cloning library " - f"(default: {VOICES_DIR}; asked for when omitted " - "in the TUI)") - parser.add_argument("--output", type=Path, default=None, - help="Output path for server.json (default: " - "server.json inside the audio.cpp checkout; an " - "existing file is overwritten only with --force " - "or a TUI confirm)") - parser.add_argument("--clone", action="store_true", - help="Non-interactive: clone audio.cpp into " - "./app/audio.cpp when no checkout is found") - parser.add_argument("--families", type=str, default=None, - help="Comma-separated model families to host, as named " - "in the audio.cpp catalog (e.g. " - "qwen3_tts,higgs_audio_tts). Required in a " - "non-interactive run; skips the family tree in " - "the TUI") - parser.add_argument("--all-packages", action="store_true", - help="Host every installable package of each selected " - "family (distinct target_directory) instead of " - "only the recommended one. Voice-design packages " - "are hosted with task 'vdes'") - parser.add_argument("--host", type=str, default=None, - help="Bind host for the server (default: 127.0.0.1)") - parser.add_argument("--port", type=int, default=None, - help="Port for the server (default: the port in " - "AUDIOCPP_API_URL from app/converter/config.py)") - parser.add_argument("--backend", choices=BACKENDS, default=None, - help="Inference backend recorded in server.json " - "(default: auto-detected from the checkout's " - "build/ directory, else cuda)") - parser.add_argument("--build-backend", choices=BACKENDS, default=None, - help="Build audiocpp_server for this backend when it " - "is not built yet, and use it in server.json") - parser.add_argument("--whisper-model", type=str, default="base", - help="Whisper model size for transcription " - "(default: base)") - parser.add_argument("--force", action="store_true", - help="Overwrite the output file (and prompt_text) " - "without prompting; in the TUI, start the " - "wizard fresh instead of loading the existing " - "server.json") - parser.add_argument("--download", action="store_true", - help="Run model_manager_v2.py install for each hosted " - "model automatically (default: print the commands " - "only)") - parser.add_argument("--no-sync-port", action="store_true", - help="Do not rewrite AUDIOCPP_API_URL in " - "app/converter/config.py when --port differs") - parser.add_argument("--no-sync-model-ids", action="store_true", - help="Do not rewrite AUDIOCPP_MODEL_ID/" - "AUDIOCPP_CLONE_MODEL_ID for a single-entry server") - return parser - - -def detect() -> BackendStatus: - """Detect how far audio.cpp is set up, plus the command to start it.""" - checkout = find_local_checkout() - details: List[str] = [] - launch = "" - if checkout is None: - # No local checkout: only a remote server can make this usable. - remote = _detect_remote() - return BackendStatus("audiocpp", "audio.cpp", installed=False, - configured=False, running=remote[0], - remote=remote[0], remote_urls=remote[1], - details=["not cloned — run setup to clone " - "./app/audio.cpp"]) - details.append(f"checkout: {checkout}") - binary = find_audiocpp_server_bin(checkout) - built = binary is not None - if built: - details.append(f"built: {binary}") - else: - details.append("not built — run setup to build audiocpp_server") - server_json = checkout / "server.json" - configured = server_json.exists() - specs: List[ServerSpec] = [] - missing = missing_model_entries(server_json) if configured else [] - if configured: - details.append(f"config: {server_json}") - if missing: - # The config references model files that are not on disk; a - # conversion would fail at model-load time, so say so now. - details.extend(model_install_hints(checkout, missing)) - if built: - # Spawned from the checkout: audiocpp_server discovers - # model_specs/<family>.json relative to its working directory. - specs = [ServerSpec( - "audiocpp", config.AUDIOCPP_API_URL, - [str(binary), "--config", str(server_json)], - cwd=checkout, identity=probe.IDENTITY_AUDIOCPP)] - else: - launch = (f"cd {checkout} && ./build/<platform>-<backend>-release" - f"/bin/audiocpp_server --config {server_json}") - else: - details.append("no server.json — run setup to configure models") - if specs: - launch = format_launch_hint(specs) - managed = servers.manages(specs) - remote_running, remote_urls = _detect_remote(managed) - # A more specific "part-way set up" label than unavailable/installed: - # cloned but never built, or built but not configured. - partial = "" - if not built: - partial = "downloaded (not built)" - elif not configured: - partial = "built (not configured)" - return BackendStatus("audiocpp", "audio.cpp", installed=built, - configured=configured, - running=managed or remote_running, - details=details, launch_hint=launch, - servers=specs, managed=managed, - remote=remote_running, remote_urls=remote_urls, - models_missing=bool(missing), partial=partial) - - -def _detect_remote(managed: bool = False) -> Tuple[bool, dict]: - """Detect an externally-run audiocpp_server at the remote URL. - - Returns ``(running, {spec_name: url})``. The remote URL is probed only - when configured (non-empty); a server answering there is ignored when it - is this tool's own managed server (remote URL == local URL and our pid is - still alive) — that instance is already reported as "[local]". - """ - url = (config.AUDIOCPP_REMOTE_URL or "").strip() - if not url: - return False, {} - if managed and probe.same_endpoint(url, config.AUDIOCPP_API_URL): - return False, {} - if probe.identify_server(url) == probe.IDENTITY_AUDIOCPP: - return True, {"audiocpp": url} - return False, {} - - -def main() -> int: - parser = build_parser() - args = parser.parse_args() - - if args.input_dir is not None and not args.input_dir.is_dir(): - parser.error( - f"WAV directory not found: {args.input_dir}\n" - f" (resolved from the current working directory: " - f"{Path.cwd()})\n" - " --wavs must be a directory containing the .wav " - "reference files to use as voice cloning presets") - - if _interactive(): - return run_tui(args, parser) - - # Non-interactive (no terminal, or all flags supplied): flag-only path. - settings = _collect_from_flags(args, parser) - if settings is None: - return 1 - return _execute(settings, args) - - -if __name__ == "__main__": - sys.exit(main()) |
