aboutsummaryrefslogtreecommitdiff
path: root/app/backends/audiocpp
diff options
context:
space:
mode:
authorhistoria <historiavg@proton.me>2026-08-26 02:25:55 -0400
committerhistoria <historiavg@proton.me>2026-08-26 02:25:55 -0400
commit8b5c8697740ff415cf7f1d03c9fb5a8c8851d420 (patch)
tree28c0323c54c896af5f89fb34b89a62e0fe0df291 /app/backends/audiocpp
parentacbd9ff2c91182d96c57ffb57bee6e9b3fcbcbd4 (diff)
downloadtts-audiobook-generator-8b5c8697740ff415cf7f1d03c9fb5a8c8851d420.tar.gz
refactor: audiocpp.py setup flow
Diffstat (limited to 'app/backends/audiocpp')
-rw-r--r--app/backends/audiocpp/__init__.py107
-rw-r--r--app/backends/audiocpp/__main__.py8
-rw-r--r--app/backends/audiocpp/build.py339
-rw-r--r--app/backends/audiocpp/catalog.py293
-rw-r--r--app/backends/audiocpp/configsync.py163
-rw-r--r--app/backends/audiocpp/constants.py32
-rw-r--r--app/backends/audiocpp/models.py384
-rw-r--r--app/backends/audiocpp/patches/ggml-top-k-cuda-iterator.patch10
-rw-r--r--app/backends/audiocpp/remote.py59
-rw-r--r--app/backends/audiocpp/status.py90
-rw-r--r--app/backends/audiocpp/voices.py146
-rw-r--r--app/backends/audiocpp/wizard.py1081
12 files changed, 2712 insertions, 0 deletions
diff --git a/app/backends/audiocpp/__init__.py b/app/backends/audiocpp/__init__.py
new file mode 100644
index 0000000..84166af
--- /dev/null
+++ b/app/backends/audiocpp/__init__.py
@@ -0,0 +1,107 @@
+"""audio.cpp backend — setup wizard, server config, model management.
+
+A package of focused modules behind one import surface: everything public
+is re-exported here, so callers (the hub, the CLI, tests) keep using
+``backends.audiocpp.<name>`` regardless of which module implements it.
+
+Modules:
+ constants shared constants (backends, tasks, checkout location)
+ catalog the model_specs catalog + server.json building/selections
+ models install state on disk, missing-model guidance, downloads
+ voices reference-.wav transcription planning and execution
+ configsync app/converter/config.py + server.json port/id/backend sync
+ build checkout lifecycle: ggml patches, binary build, uninstall
+ remote querying a running server for its models/voices
+ status detect() for the hub's backend menu
+ wizard the TUI wizard and the CLI entry points
+"""
+
+from .constants import (
+ AUDIOCPP_DIR_NAME,
+ AUDIOCPP_GIT_URL,
+ BACKENDS,
+ DEFAULT_HOST,
+ FALLBACK_PORT,
+ PATCH_DIR,
+ TASK_TTS,
+ TASK_VDES,
+)
+from .catalog import (
+ _backend_options,
+ _default_package,
+ detect_backend,
+ is_design_package,
+ load_model_catalog,
+ package_dir_options,
+ build_model_entry,
+ build_server_config,
+ load_server_config,
+ server_config_selections,
+)
+from .models import (
+ delete_model_files,
+ hand_install_guidance,
+ install_models,
+ installed_model_entries,
+ missing_model_entries,
+ missing_model_install_guidance,
+ model_install_hints,
+ unused_installed_entries,
+)
+from .voices import (
+ print_empty_transcript_warning,
+ transcribe_wav_dir,
+)
+from .configsync import (
+ config_port,
+ update_config_api_url_port,
+ update_config_model_ids,
+ update_server_backend,
+ update_server_config_port,
+)
+from .build import (
+ apply_ggml_patches,
+ build_audiocpp,
+ find_build_script,
+ find_local_checkout,
+ find_audiocpp_server_bin,
+ uninstall,
+)
+from .remote import fetch_server_models, fetch_server_voices
+from .wizard import (
+ build_parser,
+ build_screen,
+ main,
+ run_tui,
+ setup_screen,
+)
+from .status import detect
+
+__all__ = [
+ # constants
+ "AUDIOCPP_DIR_NAME", "AUDIOCPP_GIT_URL", "BACKENDS", "DEFAULT_HOST",
+ "FALLBACK_PORT", "PATCH_DIR", "TASK_TTS", "TASK_VDES",
+ # catalog
+ "detect_backend", "load_model_catalog", "is_design_package",
+ "package_dir_options", "build_model_entry", "build_server_config",
+ "load_server_config", "server_config_selections",
+ # models
+ "missing_model_entries", "installed_model_entries",
+ "unused_installed_entries", "delete_model_files", "install_models",
+ "hand_install_guidance", "missing_model_install_guidance",
+ "model_install_hints",
+ # voices
+ "transcribe_wav_dir", "print_empty_transcript_warning",
+ # configsync
+ "config_port", "update_config_api_url_port",
+ "update_config_model_ids", "update_server_config_port",
+ "update_server_backend",
+ # build
+ "find_local_checkout", "find_audiocpp_server_bin", "find_build_script",
+ "apply_ggml_patches", "build_audiocpp", "uninstall",
+ # remote
+ "fetch_server_models", "fetch_server_voices",
+ # wizard / status
+ "setup_screen", "build_screen", "run_tui", "build_parser", "detect",
+ "main",
+]
diff --git a/app/backends/audiocpp/__main__.py b/app/backends/audiocpp/__main__.py
new file mode 100644
index 0000000..457cddc
--- /dev/null
+++ b/app/backends/audiocpp/__main__.py
@@ -0,0 +1,8 @@
+"""Direct CLI execution: ``python -m backends.audiocpp`` (from app/)."""
+
+import sys
+
+from backends.audiocpp import main
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/app/backends/audiocpp/build.py b/app/backends/audiocpp/build.py
new file mode 100644
index 0000000..a4b307a
--- /dev/null
+++ b/app/backends/audiocpp/build.py
@@ -0,0 +1,339 @@
+"""Checkout lifecycle: clone location, ggml patches, binary build, uninstall."""
+
+import contextlib
+import io
+import re
+import shlex
+import shutil
+from datetime import datetime
+from pathlib import Path
+from typing import List, Optional
+
+from backends import common, servers
+from backends.common import APP_DIR
+from .catalog import _BACKEND_TOKEN_RE
+from .constants import (
+ AUDIOCPP_DIR_NAME,
+ AUDIOCPP_GIT_URL,
+ PATCH_DIR,
+)
+
+def uninstall(*, emit=None, cancel=None) -> int:
+ """Remove the audio.cpp backend entirely: stop its server, delete the checkout.
+
+ The checkout (``app/audio.cpp``) holds the built binary, the downloaded
+ models, and the server.json, so removing the directory uninstalls the
+ backend. A running server this tool started is stopped first
+ (best-effort).
+
+ EMIT is accepted for registry symmetry with the other backends but is
+ unused here — this uninstall has no subprocess phase, and its prints are
+ captured by the task view when run in the TUI. CANCEL is a
+ ``threading.Event`` honored between phases only (after the server has
+ been stopped, before the checkout is deleted), so a started phase always
+ completes and the uninstall never tears halfway. Returns the exit code
+ (130 when cancelled before a remaining phase).
+ """
+ # Only stop when a pid file exists: without one this tool never
+ # started the server, so the "not started by this tool" notice would
+ # be uninstall-time noise.
+ if servers.pid_for("audiocpp") is not None:
+ servers.stop("audiocpp")
+ if common.cancel_requested(cancel):
+ return 130
+ checkout = find_local_checkout()
+ if checkout is None:
+ print("[INFO] No audio.cpp checkout to remove.")
+ return 0
+ print(f"[INFO] Removing audio.cpp checkout {checkout}...")
+ shutil.rmtree(checkout, ignore_errors=True)
+ print("[OK] audio.cpp removed.")
+ return 0
+
+
+def find_local_checkout() -> Optional[Path]:
+ """Return the managed audio.cpp checkout at ``app/audio.cpp``.
+
+ Returns the path only when it contains a ``model_specs`` directory;
+ the checkout is installed there by the setup wizard and nowhere else.
+ """
+ try:
+ resolved = (APP_DIR / AUDIOCPP_DIR_NAME).resolve()
+ except OSError:
+ return None
+ if (resolved / "model_specs").is_dir():
+ return resolved
+ return None
+
+
+def find_audiocpp_server_bin(audiocpp_dir: Path) -> Optional[Path]:
+ """Return the built audiocpp_server binary, or None when not built.
+
+ Scans ``audiocpp_dir/build/*`` for a build directory containing
+ ``bin/audiocpp_server`` (``.exe`` allowed on Windows). When several
+ builds exist the first (alphabetical) is returned.
+ """
+ build_root = audiocpp_dir / "build"
+ if not build_root.is_dir():
+ return None
+ try:
+ build_dirs = sorted(build_root.iterdir(),
+ key=lambda p: p.name.lower())
+ except OSError:
+ return None
+ for build_dir in build_dirs:
+ if not build_dir.is_dir():
+ continue
+ for name in ("audiocpp_server", "audiocpp_server.exe"):
+ server = build_dir / "bin" / name
+ if server.exists():
+ return server
+ return None
+
+
+def built_server_binary(audiocpp_dir: Path, backend: str) -> Optional[Path]:
+ """Return the built audiocpp_server for BACKEND, or None.
+
+ Like ``find_audiocpp_server_bin`` but limited to build directories whose
+ name carries the BACKEND token (``-cuda-``, ``-vulkan-``, ``-hip-``,
+ ``-cpu-``; ``-metal-`` counts as ``cpu``). A checkout with builds for
+ several backends is asked which one to use without re-offering a build
+ for a backend that is already built.
+ """
+ build_root = audiocpp_dir / "build"
+ if not build_root.is_dir():
+ return None
+ try:
+ build_dirs = sorted(build_root.iterdir(),
+ key=lambda p: p.name.lower())
+ except OSError:
+ return None
+ for build_dir in build_dirs:
+ if not build_dir.is_dir():
+ continue
+ match = _BACKEND_TOKEN_RE.search(build_dir.name.lower())
+ if not match:
+ continue
+ token = "cpu" if match.group(1) == "metal" else match.group(1)
+ if token != backend:
+ continue
+ for name in ("audiocpp_server", "audiocpp_server.exe"):
+ server = build_dir / "bin" / name
+ if server.exists():
+ return server
+ return None
+
+
+def find_build_script(audiocpp_dir: Path) -> Optional[Path]:
+ """Return the audio.cpp build helper script to run, or None.
+
+ Prefers ``scripts/build_linux.sh``; otherwise the first
+ ``scripts/build_*.sh`` it finds. (Windows ``.bat`` scripts are not run
+ automatically — build manually there.)
+ """
+ scripts = audiocpp_dir / "scripts"
+ if not scripts.is_dir():
+ return None
+ preferred = scripts / "build_linux.sh"
+ if preferred.exists():
+ return preferred
+ try:
+ candidates = sorted(scripts.glob("build_*.sh"),
+ key=lambda p: p.name.lower())
+ except OSError:
+ return None
+ return candidates[0] if candidates else None
+
+
+GGML_PATCHES = [
+ {
+ "file": "ggml-top-k-cuda-iterator.patch",
+ "target": "external/ggml/src/ggml-cuda/top-k.cu",
+ "marker": r"#\s*include\s*<cuda/iterator>",
+ "label": "top-k.cu: add #include <cuda/iterator> (CCCL 3.x build fix)",
+ },
+]
+
+
+def apply_ggml_patches(audiocpp_dir: Path, *, emit=None, cancel=None) -> int:
+ """Apply the shipped ggml build patches to an audio.cpp checkout.
+
+ Idempotent: a patch whose marker already matches its target is skipped
+ (it is either already applied, or the fork re-vendored a fixed ggml). A
+ patch that no longer applies because the vendored file changed shape is a
+ loud, non-interactive failure — the build is aborted so the user
+ re-evaluates the patch instead of hitting a known nvcc break minutes
+ later. Returns 0 when every patch is applied or already present, 1 on
+ drift, 130 when cancelled.
+ """
+ for patch in GGML_PATCHES:
+ if cancel is not None and cancel.is_set():
+ return 130
+ target = audiocpp_dir / patch["target"]
+ if not target.is_file():
+ print(f"[INFO] {patch['file']}: target {patch['target']} not "
+ f"present in this checkout; skipping")
+ continue
+ try:
+ text = target.read_text(encoding="utf-8", errors="ignore")
+ except OSError as exc:
+ print(f"[WARNING] {patch['file']}: could not read {target}: "
+ f"{exc}; skipping")
+ continue
+ if re.search(patch["marker"], text):
+ print(f"[OK] {patch['file']}: fix already present, skipping")
+ continue
+ patch_path = PATCH_DIR / patch["file"]
+ if not patch_path.is_file():
+ print(f"[ERROR] {patch['file']}: patch file not found at "
+ f"{patch_path}; cannot apply")
+ return 1
+ check_argv = ["git", "-C", str(audiocpp_dir), "apply", "--check",
+ "--whitespace=nowarn", str(patch_path)]
+ check_rc = common.run_console_subprocess(
+ check_argv, emit=emit, cancel=cancel)
+ if check_rc == 130 or (cancel is not None and cancel.is_set()):
+ return 130
+ if check_rc != 0:
+ print(f"[ERROR] {patch['file']}: no longer applies to "
+ f"{patch['target']} (git apply --check exit {check_rc}). "
+ f"The audio.cpp fork's vendored ggml changed shape and "
+ f"still lacks the fix. Re-evaluate {patch_path}: "
+ f"regenerate the patch, or drop this entry if the fork "
+ f"now ships the fix.")
+ return 1
+ apply_argv = ["git", "-C", str(audiocpp_dir), "apply",
+ "--whitespace=nowarn", str(patch_path)]
+ rc = common.run_console_subprocess(
+ apply_argv, emit=emit, cancel=cancel)
+ if rc == 130 or (cancel is not None and cancel.is_set()):
+ return 130
+ if rc != 0:
+ print(f"[ERROR] {patch['file']}: git apply failed (exit {rc})")
+ return rc
+ print(f"[OK] {patch['file']}: applied ({patch['label']})")
+ return 0
+
+
+def build_audiocpp(audiocpp_dir: Path, backend: str, *,
+ emit=None, cancel=None) -> int:
+ """Build audiocpp_server for BACKEND, streaming output.
+
+ With EMIT None the build script runs on the console (inherits the
+ terminal); with EMIT given (the in-TUI task view) its output streams line
+ by line to EMIT so the view can show progress, and CANCEL aborts it.
+
+ On the EMIT (TUI) path the build output is also tee'd to
+ ``app/logs/audiocpp_build_<timestamp>.log`` so it survives the curses
+ session; when the build fails (and was not cancelled) a post-TUI notice
+ with the copy-pastable command and the log path is queued for the console
+ (see ``backends.common.record_post_tui_notice``).
+
+ Returns the build script's exit code (non-zero when the script is
+ missing).
+ """
+ script = find_build_script(audiocpp_dir)
+ if script is None:
+ message = (f"[ERROR] No build script found in {audiocpp_dir}/scripts; "
+ "build audiocpp_server manually (see the audio.cpp README)")
+ print(message)
+ if emit is not None:
+ common.record_post_tui_notice(message)
+ return 1
+ argv = ["sh", str(script), "--backend", backend, "--target",
+ "audiocpp_server", "--deployment-build"]
+ command = f"cd {audiocpp_dir} && {shlex.join(argv)}"
+ if emit is None:
+ print(f"[INFO] Building audiocpp_server for {backend} ({command})...")
+ patch_rc = apply_ggml_patches(audiocpp_dir, cancel=cancel)
+ if patch_rc == 130 or (cancel is not None and cancel.is_set()):
+ return 130
+ if patch_rc != 0:
+ print("[ERROR] ggml build patches could not be applied; "
+ "aborting audiocpp_server build. See the messages above "
+ "and re-evaluate app/backends/patches/.")
+ return patch_rc
+ return common.run_console_subprocess(argv, cwd=audiocpp_dir)
+ return _build_audiocpp_tui(emit, cancel, argv, command, audiocpp_dir)
+
+
+def _build_audiocpp_tui(emit, cancel, argv: List[str], command: str,
+ audiocpp_dir: Path) -> int:
+ """Run the build on the TUI path: tee output to a log file.
+
+ The ggml patch step runs first, inside the same log: every emitted
+ line (patch status, build output) is also written (and flushed) to
+ ``app/logs/audiocpp_build_<timestamp>.log``. On failure a summary (the
+ copy-pastable COMMAND and the log path) is emitted into the TUI,
+ written to the log, and queued as a post-TUI console notice. A
+ cancelled build (CANCEL set) is not reported as a failure, but its
+ partial output stays in the log file.
+ """
+ log_path = common.LOG_DIR / (
+ f"audiocpp_build_{datetime.now():%Y%m%d_%H%M%S}.log")
+ log_path.parent.mkdir(parents=True, exist_ok=True)
+ log_handle = log_path.open("w", encoding="utf-8")
+
+ def tee(line: str) -> None:
+ log_handle.write(line + "\n")
+ log_handle.flush()
+ emit(line)
+
+ class _TeeWriter(io.TextIOBase):
+ """Route print() output from the patch step into the log too."""
+
+ def write(self, s: str) -> int:
+ for line in s.splitlines():
+ if line:
+ tee(line)
+ return len(s)
+
+ try:
+ with contextlib.redirect_stdout(_TeeWriter()):
+ patch_rc = apply_ggml_patches(audiocpp_dir, emit=tee,
+ cancel=cancel)
+ if patch_rc == 130 or (cancel is not None and cancel.is_set()):
+ return 130
+ if patch_rc != 0:
+ notice = ("[ERROR] ggml build patches could not be applied; "
+ "aborting audiocpp_server build. See the messages "
+ "above and re-evaluate app/backends/patches/.")
+ tee(notice)
+ common.record_post_tui_notice(notice)
+ return patch_rc
+ tee(f"[INFO] Building audiocpp_server ({command})...")
+ rc = common.run_console_subprocess(
+ argv, cwd=audiocpp_dir, emit=tee, cancel=cancel)
+ if rc != 0 and (cancel is None or not cancel.is_set()):
+ notice = (f"[ERROR] audio.cpp build failed (exit code {rc}).\n"
+ f" Build log: {log_path}\n"
+ f" Troubleshoot by re-running this command:\n"
+ f" {command}")
+ for line in notice.splitlines():
+ tee(line)
+ common.record_post_tui_notice(notice)
+ finally:
+ log_handle.close()
+ return rc
+
+
+def _print_launch_hint(audiocpp_dir: Path, output_path: Path) -> None:
+ """Print remediation when audiocpp_server is missing (troubleshooting).
+
+ The hub starts and stops the server itself, so a working install gets
+ no manual launch instructions. When no binary was built, though, the
+ user needs to know how to build and run it by hand. The commands are
+ prefixed with ``cd <checkout> &&`` because the server discovers
+ model_specs/<family>.json relative to its working directory.
+ """
+ if find_audiocpp_server_bin(audiocpp_dir) is not None:
+ return
+ print("\n[INFO] audiocpp_server binary not found. Build it first, e.g.:")
+ script = find_build_script(audiocpp_dir)
+ if script is not None:
+ print(f" sh {script} --backend <cuda|vulkan|hip|cpu> "
+ "--target audiocpp_server --deployment-build")
+ print(f" then run: cd {audiocpp_dir} && ./build/<platform>-<backend>"
+ f"-release/bin/audiocpp_server --config {output_path}")
+
+
diff --git a/app/backends/audiocpp/catalog.py b/app/backends/audiocpp/catalog.py
new file mode 100644
index 0000000..c672232
--- /dev/null
+++ b/app/backends/audiocpp/catalog.py
@@ -0,0 +1,293 @@
+"""The model_specs catalog, server.json building and selection views."""
+
+import json
+import re
+from pathlib import Path
+from typing import Dict, List, Optional, Set, Tuple
+
+from .. import common
+from .constants import (
+ BACKENDS,
+ DEFAULT_HOST,
+ FALLBACK_PORT,
+ TASK_TTS,
+)
+
+DESIGN_PACKAGE_RE = re.compile(r"voice[\s_\-]?design", re.IGNORECASE)
+
+
+_BACKEND_DESCRIPTIONS = (
+ ("cuda", "NVIDIA GPUs (fastest)"),
+ ("vulkan", "cross-vendor GPU"),
+ ("hip", "AMD GPUs"),
+ ("cpu", "no GPU required"),
+)
+
+
+def _backend_options(detected: Optional[str] = None
+ ) -> Tuple[List[Tuple[str, str]], int]:
+ """Build the aligned backend menu options and the default index.
+
+ The backend names are padded to a common width so the ``-`` dashes
+ before the descriptions line up. When DETECTED matches one of the
+ options, that option gets ``[auto-detected]`` appended and is the
+ default (cursor/start) selection; otherwise the first option is the
+ default as before. Returns (options, default_index).
+ """
+ width = max(len(name) for name, _ in _BACKEND_DESCRIPTIONS)
+ options: List[Tuple[str, str]] = []
+ default_index = 0
+ for index, (name, desc) in enumerate(_BACKEND_DESCRIPTIONS):
+ label = f"{name.ljust(width)} - {desc}"
+ if detected == name:
+ label += " [auto-detected]"
+ default_index = index
+ options.append((label, name))
+ return options, default_index
+
+
+_BACKEND_TOKEN_RE = re.compile(r"-(cuda|vulkan|hip|cpu|metal)(?:-|$)")
+
+
+def detect_backend(audiocpp_dir: Path) -> Optional[str]:
+ """Best-effort detection of the backend audiocpp_server was built for.
+
+ Scans ``audiocpp_dir/build/*`` for build directories that contain a
+ built ``bin/audiocpp_server`` (``.exe`` allowed on Windows) and reads
+ the backend token out of the directory name (``-cuda-``, ``-vulkan-``,
+ ``-hip-`` or ``-cpu-``; ``-metal-`` is mapped to ``cpu``). Returns the
+ backend only when exactly one distinct backend was built, so a checkout
+ with builds for several backends does not silently pick one. Returns
+ None when there is no ``build/`` directory, no built server, or more
+ than one distinct backend.
+ """
+ build_root = audiocpp_dir / "build"
+ if not build_root.is_dir():
+ return None
+ backends: Set[str] = set()
+ try:
+ build_dirs = sorted(build_root.iterdir(),
+ key=lambda p: p.name.lower())
+ except OSError:
+ return None
+ for build_dir in build_dirs:
+ if not build_dir.is_dir():
+ continue
+ server = build_dir / "bin" / "audiocpp_server"
+ if not server.exists():
+ server_exe = build_dir / "bin" / "audiocpp_server.exe"
+ if not server_exe.exists():
+ continue
+ match = _BACKEND_TOKEN_RE.search(build_dir.name.lower())
+ if not match:
+ continue
+ token = match.group(1)
+ backends.add("cpu" if token == "metal" else token)
+ if len(backends) == 1:
+ return next(iter(backends))
+ return None
+
+
+def _default_package(packages: List[dict]) -> Optional[dict]:
+ """Pick the default package from a list of packages.
+
+ Prefers the package flagged ``default: true``, then the first GGUF
+ package, then the first package overall. Returns None for an empty list.
+ """
+ if not packages:
+ return None
+ for package in packages:
+ if package.get("default"):
+ return package
+ for package in packages:
+ if package.get("format") == "gguf":
+ return package
+ return packages[0]
+
+
+def load_model_catalog(audiocpp_dir: Path) -> List[dict]:
+ """Read model_specs/*.json and return the TTS-capable families.
+
+ Each returned entry has: family, display_name, description, languages,
+ clone_capable, packages (the full list from the spec), install_id
+ (recommended package id), and default_path (``models/<target_directory>``).
+ All families are treated equally and listed in alphabetical order by
+ display name.
+ """
+ specs_dir = audiocpp_dir / "model_specs"
+ if not specs_dir.is_dir():
+ raise NotADirectoryError(
+ f"{audiocpp_dir} has no model_specs/ directory; re-run setup "
+ "to refresh the audio.cpp checkout")
+ entries: List[dict] = []
+ for spec_path in sorted(specs_dir.glob("*.json")):
+ try:
+ spec = json.loads(spec_path.read_text(encoding="utf-8"))
+ except (OSError, ValueError):
+ continue
+ tasks = spec.get("tasks") or []
+ if "tts" not in tasks and spec.get("category") != "tts":
+ continue
+ family = spec.get("family") or spec_path.stem
+ packages = spec.get("packages") or []
+ package = _default_package(packages)
+ if package is None:
+ # No installable package: skip (cannot be hosted from a path).
+ continue
+ target_directory = package.get("target_directory") or family
+ languages = spec.get("languages") or []
+ display_name = spec.get("display_name") or family
+ description = spec.get("description") or ""
+ entries.append({
+ "family": family,
+ "display_name": display_name,
+ "description": description,
+ "languages": languages,
+ "tasks": list(tasks),
+ "clone_capable": "clone" in tasks,
+ "packages": packages,
+ "install_id": package.get("id") or family,
+ "default_path": f"models/{target_directory}",
+ })
+
+ # All families are treated equally: alphabetical by display name.
+ entries.sort(key=lambda entry: entry["display_name"].lower())
+ return entries
+
+
+def is_design_package(package: dict) -> bool:
+ """Return True when a package's name marks it a voice-design model.
+
+ audio.cpp voice-design packages (whose id, display name, or target
+ directory mentions "voice design") are the only packages that must be
+ hosted with task "vdes"; their role is not in the schema, only in those
+ strings, so it is detected from them.
+ """
+ text = " ".join(str(package.get(key, ""))
+ for key in ("id", "display_name", "target_directory"))
+ return bool(DESIGN_PACKAGE_RE.search(text))
+
+
+def package_dir_options(entry: dict) -> List[dict]:
+ """Return one option per distinct target_directory of a family's packages.
+
+ Each option is a dict with: target_directory, install_id (the recommended
+ package id inside that directory), design (voice-design package flag), and
+ recommended (whether it holds the family's default package). Precisions
+ that share a directory (q8_0/bf16/...) collapse to a single option.
+ """
+ packages = entry.get("packages") or []
+ default_pkg = _default_package(packages)
+ default_dir = (default_pkg or {}).get("target_directory") or entry["family"]
+ by_dir: Dict[str, List[dict]] = {}
+ order: List[str] = []
+ for package in packages:
+ directory = package.get("target_directory") or entry["family"]
+ if directory not in by_dir:
+ by_dir[directory] = []
+ order.append(directory)
+ by_dir[directory].append(package)
+ options: List[dict] = []
+ for directory in order:
+ package = _default_package(by_dir[directory])
+ options.append({
+ "target_directory": directory,
+ "install_id": (package or {}).get("id") or directory,
+ "design": is_design_package(package or {}),
+ "recommended": directory == default_dir,
+ })
+ # Put the recommended package first for a friendlier checklist.
+ options.sort(key=lambda opt: not opt["recommended"])
+ return options
+
+
+def build_model_entry(family: str, model_id: str, model_path: str,
+ task: str = TASK_TTS) -> dict:
+ """Assemble one server.json model entry.
+
+ ``task`` defaults to "tts"; voice design packages are hosted with
+ "vdes" so the server runs its design session for speech requests
+ (audiobook.py then requires --instructions with that entry).
+ """
+ return {
+ "id": model_id,
+ "family": family,
+ "path": model_path,
+ "task": task,
+ "mode": "offline",
+ }
+
+
+def build_server_config(host: str, port: int, backend: str, lazy_load: bool,
+ model_entries: List[dict],
+ voice_dir: Optional[str] = None) -> dict:
+ """Assemble the server.json document.
+
+ ``voice_dir`` is a server-level cloning voice library; when set, every
+ hosted clone-capable family can use its voices with ``--voice``.
+ """
+ config_doc = {
+ "host": host,
+ "port": port,
+ "backend": backend,
+ "lazy_load": lazy_load,
+ "models": model_entries,
+ }
+ if voice_dir:
+ config_doc["voice_dir"] = voice_dir
+ return config_doc
+
+
+def load_server_config(server_json: Path) -> Optional[dict]:
+ """Read server.json into a dict, or None when it cannot be used.
+
+ Returns None for a missing file, unreadable content, or a non-dict
+ document. Used by the wizard's modify flow to pre-fill its screens
+ from an existing config instead of prompting to overwrite it.
+ """
+ if not server_json.exists():
+ return None
+ try:
+ data = json.loads(server_json.read_text(encoding="utf-8"))
+ except (OSError, ValueError):
+ return None
+ if not isinstance(data, dict):
+ return None
+ return data
+
+
+def server_config_selections(server_config: dict,
+ catalog: List[dict]
+ ) -> Tuple[Dict[str, List[str]],
+ Dict[Tuple[str, str], str]]:
+ """Map an existing server.json's models back to catalog selections.
+
+ Returns ``(selected_dirs, tasks)``: ``selected_dirs`` maps a catalog
+ family to the target directories it hosts (``models/<target>`` paths
+ with the ``models/`` prefix stripped, in server.json order), and
+ ``tasks`` maps ``(family, target_directory)`` to the entry's task
+ (``"tts"`` or ``"vdes"``) so the wizard can preserve how design
+ packages were hosted. Entries whose family is not in the CATALOG are
+ ignored — the wizard cannot offer them again.
+ """
+ families = {entry["family"] for entry in catalog}
+ selected_dirs: Dict[str, List[str]] = {}
+ tasks: Dict[Tuple[str, str], str] = {}
+ for entry in server_config.get("models") or []:
+ if not isinstance(entry, dict):
+ continue
+ family = entry.get("family")
+ if not isinstance(family, str) or family not in families:
+ continue
+ path = entry.get("path")
+ if not isinstance(path, str):
+ continue
+ target = path[len("models/"):] if path.startswith("models/") else path
+ if family not in selected_dirs:
+ selected_dirs[family] = []
+ if target not in selected_dirs[family]:
+ selected_dirs[family].append(target)
+ tasks[(family, target)] = str(entry.get("task") or TASK_TTS)
+ return selected_dirs, tasks
+
+
diff --git a/app/backends/audiocpp/configsync.py b/app/backends/audiocpp/configsync.py
new file mode 100644
index 0000000..0a161c0
--- /dev/null
+++ b/app/backends/audiocpp/configsync.py
@@ -0,0 +1,163 @@
+"""Keep app/converter/config.py and server.json in sync with setup choices."""
+
+import json
+import re
+import urllib.parse
+from pathlib import Path
+from typing import Optional
+
+from backends import common
+from backends.common import CONFIG_PATH, url_with_port
+from converter import config
+from . import build
+from .constants import FALLBACK_PORT
+
+def config_port() -> int:
+ """Return the port of AUDIOCPP_API_URL in app/converter/config.py."""
+ try:
+ return urllib.parse.urlsplit(config.AUDIOCPP_API_URL).port or FALLBACK_PORT
+ except ValueError:
+ return FALLBACK_PORT
+
+
+def update_config_api_url_port(port: int, config_path: Optional[Path] = None) -> bool:
+ """Rewrite the port inside AUDIOCPP_API_URL in app/converter/config.py.
+
+ Reads the configured URL from the file (not from the imported module,
+ which a long hub session can leave behind), swaps its port for PORT,
+ and writes it back through ``common.update_config_value`` so the
+ imported module mirrors the change immediately. Returns True when the
+ file now holds the new URL.
+ """
+ path = Path(config_path) if config_path is not None else CONFIG_PATH
+ try:
+ text = path.read_text(encoding="utf-8")
+ except OSError:
+ return False
+ match = re.search(r'(?m)^\s*AUDIOCPP_API_URL\s*=\s*"([^"]*)"', text)
+ if not match:
+ return False
+ return common.update_config_value("AUDIOCPP_API_URL",
+ url_with_port(match.group(1), port),
+ config_path=path)
+
+
+def update_server_config_port(port: int) -> bool:
+ """Rewrite the 'port' in the audio.cpp checkout's server.json.
+
+ Loads ``<checkout>/server.json``, sets its ``port`` to PORT, and
+ rewrites it with the same ``json.dump`` formatting the wizard uses.
+ Returns True when the file now carries PORT (a no-op when it already
+ does), and False when there is no checkout/server.json or the file
+ cannot be read or written.
+ """
+ checkout = build.find_local_checkout()
+ if checkout is None:
+ return False
+ server_json = checkout / "server.json"
+ if not server_json.exists():
+ return False
+ try:
+ data = json.loads(server_json.read_text(encoding="utf-8"))
+ except (OSError, ValueError):
+ return False
+ if not isinstance(data, dict):
+ return False
+ if data.get("port") == port:
+ return True
+ data["port"] = port
+ try:
+ with server_json.open("w", encoding="utf-8") as handle:
+ json.dump(data, handle, indent=2, ensure_ascii=False)
+ handle.write("\n")
+ except OSError:
+ return False
+ return True
+
+
+def update_config_model_ids(model_id: str,
+ clone_model_id: Optional[str] = None,
+ config_path: Optional[Path] = None) -> bool:
+ """Rewrite AUDIOCPP_MODEL_ID (and AUDIOCPP_CLONE_MODEL_ID when given).
+
+ Goes through ``common.update_config_value`` so the imported config
+ module mirrors the change immediately. Returns True when every named
+ key now holds its value in the file.
+ """
+ path = Path(config_path) if config_path is not None else CONFIG_PATH
+ ok = common.update_config_value("AUDIOCPP_MODEL_ID", model_id,
+ config_path=path)
+ if clone_model_id is not None:
+ ok = common.update_config_value("AUDIOCPP_CLONE_MODEL_ID",
+ clone_model_id,
+ config_path=path) and ok
+ return ok
+
+
+def _apply_port_sync(port: int, accepted: bool) -> None:
+ """Write the port into app/converter/config.py, or report when declined."""
+ if accepted:
+ if not update_config_api_url_port(port):
+ print(f"[WARNING] Could not update {CONFIG_PATH}; edit "
+ "AUDIOCPP_API_URL by hand so audiobook.py uses the "
+ "new port")
+ else:
+ print("[WARNING] Left AUDIOCPP_API_URL unchanged; audiobook.py "
+ f"will still use port {config_port()}")
+
+
+def _offer_config_model_id_sync(model_id: str, accepted: Optional[bool]) -> None:
+ """Point app/converter/config.py at a single hosted model entry.
+
+ The converter requests the model id configured in AUDIOCPP_MODEL_ID,
+ and single-model servers use the same id for the clone entry, so both
+ ids are rewritten together. ACCEPTED is True/False (apply/skip the
+ rewrite) or None when no single-entry sync applies (nothing to do).
+ """
+ if config.AUDIOCPP_MODEL_ID == model_id \
+ and config.AUDIOCPP_CLONE_MODEL_ID == model_id:
+ return
+ if accepted is None:
+ return
+ if accepted:
+ if not update_config_model_ids(model_id, model_id):
+ print(f"[WARNING] Could not update {CONFIG_PATH}; edit "
+ "AUDIOCPP_MODEL_ID and AUDIOCPP_CLONE_MODEL_ID by hand so "
+ "audiobook.py uses this model")
+ else:
+ print("[WARNING] Left the model ids unchanged; audiobook.py will "
+ f"still request model '{config.AUDIOCPP_MODEL_ID}'")
+
+
+def update_server_backend(backend: str) -> bool:
+ """Rewrite the 'backend' in the checkout's server.json, or True when none.
+
+ Sets ``backend`` to BACKEND in ``<checkout>/server.json`` (same
+ ``json.dump`` formatting as the wizard). Returns True when the file now
+ carries BACKEND, when there is no server.json (nothing to sync), or when
+ it already does; False when the file exists but cannot be read/written.
+ """
+ checkout = build.find_local_checkout()
+ if checkout is None:
+ return True
+ server_json = checkout / "server.json"
+ if not server_json.exists():
+ return True
+ try:
+ data = json.loads(server_json.read_text(encoding="utf-8"))
+ except (OSError, ValueError):
+ return False
+ if not isinstance(data, dict):
+ return False
+ if data.get("backend") == backend:
+ return True
+ data["backend"] = backend
+ try:
+ with server_json.open("w", encoding="utf-8") as handle:
+ json.dump(data, handle, indent=2, ensure_ascii=False)
+ handle.write("\n")
+ except OSError:
+ return False
+ return True
+
+
diff --git a/app/backends/audiocpp/constants.py b/app/backends/audiocpp/constants.py
new file mode 100644
index 0000000..aaa1eed
--- /dev/null
+++ b/app/backends/audiocpp/constants.py
@@ -0,0 +1,32 @@
+"""Constants shared across the audio.cpp backend modules."""
+
+import re
+from pathlib import Path
+
+DEFAULT_HOST = "127.0.0.1"
+
+
+FALLBACK_PORT = 8080
+
+
+BACKENDS = ("cuda", "vulkan", "hip", "cpu")
+
+
+TASK_TTS = "tts"
+
+
+TASK_VDES = "vdes"
+
+
+AUDIOCPP_DIR_NAME = "audio.cpp"
+
+
+AUDIOCPP_GIT_URL = "https://github.com/0xShug0/audio.cpp"
+
+
+
+# ggml build patches shipped in this repo and applied to the (gitignored)
+# audio.cpp checkout before building, so a fresh clone survives known ggml
+# build bugs the audio.cpp fork has not re-vendored yet. See
+# apply_ggml_patches() in backends.audiocpp.build.
+PATCH_DIR = Path(__file__).resolve().parent / "patches"
diff --git a/app/backends/audiocpp/models.py b/app/backends/audiocpp/models.py
new file mode 100644
index 0000000..4e6b8bb
--- /dev/null
+++ b/app/backends/audiocpp/models.py
@@ -0,0 +1,384 @@
+"""Model install state: what is on disk, what is missing, how to fetch it."""
+
+import json
+import os
+import shutil
+import sys
+import tempfile
+from pathlib import Path
+from typing import Callable, Dict, List, Optional, Set, Tuple
+
+from backends import common
+from . import catalog as _catalog
+
+def _install_models(audiocpp_dir: Path,
+ install_guidance: List[Tuple[str, str]],
+ download: bool, emit=None, cancel=None) -> int:
+ """Print and optionally run the model install commands.
+
+ One ``python <manager> install <id>`` command per hosted model (de-duped
+ by install id). When DOWNLOAD is True each command is run in the audio.cpp
+ checkout via ``subprocess`` so the models are downloaded automatically;
+ a failing install is reported as a warning and does not abort the
+ remaining downloads. When DOWNLOAD is False (or the model manager is
+ missing) the commands are only printed, copy-pasteable as before.
+
+ With EMIT given (the in-TUI task view) each download streams its output
+ to EMIT and — when the checkout's ``model_manager_v2.py`` supports it —
+ runs with ``--progress --cancel-file`` so the view can show a real byte
+ progress bar and cancel gracefully. CANCEL aborts a running download.
+
+ Returns 0 when every command succeeded (or nothing needed running),
+ 130 when cancelled, 1 when any download failed.
+ """
+ manager = audiocpp_dir / "tools" / "model_manager_v2.py"
+ seen: Set[str] = set()
+ install_ids: List[str] = []
+ for _, install_id in install_guidance:
+ if install_id not in seen:
+ seen.add(install_id)
+ install_ids.append(install_id)
+
+ supports_progress = emit is not None and _manager_supports_progress(manager)
+
+ if download and not manager.is_file():
+ print(f"[WARNING] {manager} not found; printing the install commands "
+ "instead of running them")
+ download = False
+
+ failed = False
+ for install_id in install_ids:
+ command = f"python {manager} install {install_id}"
+ if not download:
+ print(command)
+ continue
+ print(f"[INFO] Downloading {install_id}...")
+ argv = [sys.executable, str(manager), "install", install_id]
+ cancel_file: Optional[Path] = None
+ on_cancel = None
+ if supports_progress:
+ fd, cancel_path = tempfile.mkstemp(
+ prefix="audiocpp_cancel_", suffix=".cancel")
+ os.close(fd)
+ cancel_file = Path(cancel_path)
+ cancel_file.unlink() # absent = not cancelled
+ argv += ["--progress", "--cancel-file", str(cancel_file)]
+ on_cancel = cancel_file.touch
+ try:
+ rc = common.run_console_subprocess(
+ argv, cwd=str(audiocpp_dir), emit=emit, cancel=cancel,
+ on_cancel=on_cancel)
+ except OSError as exc:
+ print(f"[WARNING] Could not run {command}: {exc}")
+ rc = 1
+ finally:
+ if cancel_file is not None:
+ try:
+ cancel_file.unlink()
+ except OSError:
+ pass
+ if rc == 130 or (cancel is not None and cancel.is_set()):
+ return 130
+ if rc != 0:
+ failed = True
+ print(f"[WARNING] install {install_id} exited with code "
+ f"{rc}; the model may need to be downloaded "
+ "by hand")
+ return 1 if failed else 0
+
+
+def _manager_supports_progress(manager: Path) -> bool:
+ """True when MANAGER (model_manager_v2.py) supports --progress output.
+
+ The ``--progress``/``--cancel-file`` flags are relatively recent; an
+ older audio.cpp checkout may not have them, so probe the script source
+ once instead of failing the download with an unknown flag.
+ """
+ try:
+ text = manager.read_text(encoding="utf-8", errors="ignore")
+ except OSError:
+ return False
+ return "AUDIOCPP_PROGRESS" in text and "--cancel-file" in text
+
+
+def _decide_download(audiocpp_dir: Path,
+ model_entries: List[dict],
+ confirm: Callable[[str, bool], bool]) -> bool:
+ """Ask whether to download the selected models now.
+
+ CONFIRM asks the yes/no question (ask_bool for the line prompts, a TUI
+ confirm for the wizard). When the audio.cpp model manager is missing the
+ prompt is skipped and False is returned, so the install commands are only
+ printed rather than offered to run. The prompt is also skipped (False)
+ when every selected model is already on disk (see ``_all_models_present``),
+ so an already-configured checkout is not asked to re-download models it
+ already has.
+ """
+ manager = audiocpp_dir / "tools" / "model_manager_v2.py"
+ if not manager.is_file():
+ return False
+ if _all_models_present(audiocpp_dir, model_entries):
+ return False
+ return confirm(
+ "Automatically download the selected models with model_manager_v2.py "
+ "now?", True)
+
+
+def _build_tree_families(catalog: List[dict]) -> List[dict]:
+ """Shape the catalog into the checkbox_tree widget's family list."""
+ families: List[dict] = []
+ for entry in catalog:
+ capabilities = ["tts"]
+ if "clone" in entry["tasks"]:
+ capabilities.append("cloning")
+ if "design" in entry["tasks"]:
+ capabilities.append("design")
+ name = entry["display_name"]
+ options = []
+ for opt in _catalog.package_dir_options(entry):
+ options.append({
+ "key": opt["target_directory"],
+ "label": opt["install_id"],
+ "recommended": opt["recommended"],
+ })
+ families.append({
+ "label": name,
+ "detail": ", ".join(capabilities),
+ "options": options,
+ })
+ return families
+
+
+def _model_path_present(path: Path) -> bool:
+ """True when a server.json model path holds actual model files.
+
+ A present path is either a file (a single-model package) or a non-empty
+ directory (the usual GGUF package target directory; an empty one means a
+ download that never ran or was cleaned up halfway).
+ """
+ try:
+ if path.is_file():
+ return True
+ if path.is_dir():
+ return any(path.iterdir())
+ except OSError:
+ return False
+ return False
+
+
+def _all_models_present(audiocpp_dir: Path, model_entries: List[dict]) -> bool:
+ """True when every selected model entry's path already holds files on disk.
+
+ Paths resolve against the checkout (where model_manager_v2.py installs
+ them), honoring absolute paths. Used by the wizard to skip the
+ "Automatically download the selected models" prompt when nothing is
+ actually missing. An empty selection is treated as not-present.
+ """
+ if not model_entries:
+ return False
+ for entry in model_entries:
+ rel = entry.get("path")
+ if not isinstance(rel, str) or not rel:
+ return False
+ path = Path(rel) if Path(rel).is_absolute() else audiocpp_dir / rel
+ if not _model_path_present(path):
+ return False
+ return True
+
+
+def missing_model_entries(server_json: Path) -> List[dict]:
+ """Return the server.json model entries whose files are not on disk.
+
+ Paths resolve exactly like audiocpp_server resolves them (relative paths
+ against the server.json's directory). Each returned entry carries the
+ entry ``id`` and ``rel`` (the configured path string); used by ``detect``
+ to warn that a conversion would fail until the models are installed.
+ """
+ try:
+ data = json.loads(server_json.read_text(encoding="utf-8"))
+ except (OSError, ValueError):
+ return []
+ if not isinstance(data, dict):
+ return []
+ base = server_json.parent
+ missing: List[dict] = []
+ for entry in data.get("models") or []:
+ if not isinstance(entry, dict):
+ continue
+ rel = entry.get("path")
+ if not isinstance(rel, str) or not rel:
+ continue
+ path = Path(rel) if Path(rel).is_absolute() else base / rel
+ if _model_path_present(path):
+ continue
+ missing.append({"id": str(entry.get("id") or rel), "rel": rel})
+ return missing
+
+
+def _install_id_by_path(audiocpp_dir: Path) -> Dict[str, str]:
+ """Map ``models/<target_directory>`` -> catalog install id.
+
+ The catalog package that installs a model is derived from the
+ ``default_path`` of each TTS family; an entry whose path matches no
+ catalog package has no install id.
+ """
+ by_path: Dict[str, str] = {}
+ try:
+ for entry in _catalog.load_model_catalog(audiocpp_dir):
+ by_path[entry["default_path"]] = entry["install_id"]
+ except (NotADirectoryError, OSError):
+ pass
+ return by_path
+
+
+def installed_model_entries(server_json: Path) -> List[dict]:
+ """Return the server.json model entries whose files ARE on disk.
+
+ The complement of ``missing_model_entries``: each returned entry carries
+ the entry ``id`` and ``rel`` (the configured path string), resolved
+ exactly like ``missing_model_entries`` (relative against the server.json's
+ directory). Used by the wizard's "Delete unused models?" step to find
+ already-downloaded models that were unselected.
+ """
+ try:
+ data = json.loads(server_json.read_text(encoding="utf-8"))
+ except (OSError, ValueError):
+ return []
+ if not isinstance(data, dict):
+ return []
+ base = server_json.parent
+ installed: List[dict] = []
+ for entry in data.get("models") or []:
+ if not isinstance(entry, dict):
+ continue
+ rel = entry.get("path")
+ if not isinstance(rel, str) or not rel:
+ continue
+ path = Path(rel) if Path(rel).is_absolute() else base / rel
+ if _model_path_present(path):
+ installed.append({"id": str(entry.get("id") or rel), "rel": rel})
+ return installed
+
+
+def missing_model_install_guidance(audiocpp_dir: Path,
+ missing: List[dict]) -> List[Tuple[str, str]]:
+ """Map MISSING model entries to (display name, install id) pairs.
+
+ The install id is derived from each entry's configured path via the
+ catalog (see ``_install_id_by_path``); entries whose path matches no
+ catalog package are skipped (there is no ``model_manager_v2.py install``
+ command for them). Feeds ``_install_models`` for the "Download Missing
+ Models" action.
+ """
+ by_path = _install_id_by_path(audiocpp_dir)
+ guidance: List[Tuple[str, str]] = []
+ for item in missing:
+ install_id = by_path.get(item["rel"])
+ if install_id:
+ guidance.append((item["id"], install_id))
+ return guidance
+
+
+def model_install_hints(audiocpp_dir: Path,
+ missing: List[dict]) -> List[str]:
+ """Remediation lines for MISSING model entries (see missing_model_entries).
+
+ Maps each entry's configured path back to the catalog package that
+ installs it (``models/<target_directory>`` -> install id) so the line
+ carries the exact ``model_manager_v2.py install`` command; entries whose
+ directory matches no catalog package just name the path.
+ """
+ by_path = _install_id_by_path(audiocpp_dir)
+ hints: List[str] = []
+ for item in missing:
+ install_id = by_path.get(item["rel"])
+ hint = f"model not downloaded: {item['id']} ({item['rel']})"
+ if install_id:
+ hint += (f" — install with: python tools/model_manager_v2.py "
+ f"install {install_id}")
+ hints.append(hint)
+ return hints
+
+
+def install_models(audiocpp_dir: Path,
+ guidance: List[Tuple[str, str]],
+ emit=None, cancel=None) -> int:
+ """Download the (display name, install id) models via the helper script.
+
+ Runs ``model_manager_v2.py install`` for each de-duped install id in the
+ checkout, streaming to the console (or to EMIT, the in-TUI task view); a
+ failing install is reported as a warning and does not abort the rest.
+ Returns 0 when every download succeeded, 130 when cancelled, 1 when any
+ failed. Used by the hub's "Download Missing Models" action (see
+ ``missing_model_install_guidance`` for the mapping).
+ """
+ return _install_models(audiocpp_dir, guidance, download=True,
+ emit=emit, cancel=cancel)
+
+
+def hand_install_guidance(audiocpp_dir: Path,
+ missing: List[dict]) -> str:
+ """Explain how to install MISSING model entries by hand.
+
+ Returns a multi-line message listing each missing model's id and the
+ path its files must be placed in (``rel``, resolved against the
+ checkout). Used when the missing models cannot be mapped to
+ a ``model_manager_v2.py install`` command, so the user still knows what
+ to download and where to put it.
+ """
+ lines = [
+ "None of the missing models map to a model_manager_v2.py install "
+ "command.",
+ "Download them by hand and place the files at these paths:",
+ ]
+ for item in missing:
+ lines.append(f" {item['id']} -> {item['rel']}")
+ lines.append(f"(paths are relative to {audiocpp_dir})")
+ return "\n".join(lines)
+
+
+def unused_installed_entries(server_json: Path,
+ new_paths: Set[str]) -> List[dict]:
+ """Return installed server.json entries whose path is not in NEW_PATHS.
+
+ The already-downloaded models (see ``installed_model_entries``) that the
+ new selection does not host any more — the candidates for the wizard's
+ "Delete unused models?" prompt. Entries whose files are not on disk are
+ never listed (there is nothing to delete).
+ """
+ return [entry for entry in installed_model_entries(server_json)
+ if entry["rel"] not in new_paths]
+
+
+def delete_model_files(server_json: Path, entries: List[dict]) -> int:
+ """Remove the on-disk model files for ENTRIES ({id, rel}) from disk.
+
+ Each entry's ``rel`` is resolved exactly like the server resolves it
+ (relative against ``server_json``'s directory; absolute paths honored),
+ then removed as a directory tree or a single file. Missing entries are
+ ignored. Returns the number of paths removed. Used by the wizard's
+ "Delete unused models?" step — the regenerated server.json already only
+ lists the kept models, so no entry cleanup is needed here.
+ """
+ base = server_json.parent
+ removed = 0
+ for item in entries:
+ rel = item.get("rel")
+ if not isinstance(rel, str) or not rel:
+ continue
+ path = Path(rel) if Path(rel).is_absolute() else base / rel
+ try:
+ if not path.exists():
+ continue
+ if path.is_dir():
+ shutil.rmtree(path, ignore_errors=True)
+ else:
+ path.unlink()
+ except OSError as exc:
+ print(f"[WARNING] Could not remove {path}: {exc}")
+ continue
+ print(f"[OK] Removed unused model {path}")
+ removed += 1
+ return removed
+
+
diff --git a/app/backends/audiocpp/patches/ggml-top-k-cuda-iterator.patch b/app/backends/audiocpp/patches/ggml-top-k-cuda-iterator.patch
new file mode 100644
index 0000000..0eb89a5
--- /dev/null
+++ b/app/backends/audiocpp/patches/ggml-top-k-cuda-iterator.patch
@@ -0,0 +1,10 @@
+--- a/external/ggml/src/ggml-cuda/top-k.cu
++++ b/external/ggml/src/ggml-cuda/top-k.cu
+@@ -4,6 +4,7 @@
+ #ifdef GGML_CUDA_USE_CUB
+ # include <cub/cub.cuh>
+ # if (CCCL_MAJOR_VERSION >= 3 && CCCL_MINOR_VERSION >= 2)
+ # define CUB_TOP_K_AVAILABLE
++# include <cuda/iterator>
+ using namespace cub;
+ # endif // CCCL_MAJOR_VERSION >= 3 && CCCL_MINOR_VERSION >= 2
diff --git a/app/backends/audiocpp/remote.py b/app/backends/audiocpp/remote.py
new file mode 100644
index 0000000..42b3872
--- /dev/null
+++ b/app/backends/audiocpp/remote.py
@@ -0,0 +1,59 @@
+"""Query a running audiocpp_server for its models and voices."""
+
+import json
+import urllib.request
+from typing import Dict, List, Optional
+
+from .constants import FALLBACK_PORT
+
+def fetch_server_models(api_url: str) -> Optional[List[Dict[str, str]]]:
+ """List a running audiocpp_server's model entries via GET /v1/models.
+
+ Returns ``[{id, family, task}, ...]`` — the same shape the converter's
+ client resolves at startup — or None when URL does not answer with a
+ valid document (wrong server, still starting, older audio.cpp). Used by
+ the hub to drive the convert menus against a remote server that has no
+ local server.json describing it.
+ """
+ try:
+ with urllib.request.urlopen(
+ f"{api_url.rstrip('/')}/v1/models", timeout=10) as response:
+ payload = json.loads(response.read().decode("utf-8"))
+ except (OSError, ValueError):
+ # URLError/HTTPError/socket errors are OSErrors; a non-JSON body is
+ # a ValueError. Anything else means "not an audiocpp_server".
+ return None
+ entries = payload.get("data") if isinstance(payload, dict) else None
+ models: List[Dict[str, str]] = []
+ for entry in entries or []:
+ if isinstance(entry, dict) and entry.get("id"):
+ models.append({
+ "id": str(entry["id"]),
+ "family": str(entry.get("family") or ""),
+ "task": str(entry.get("task") or ""),
+ })
+ return models
+
+
+def fetch_server_voices(api_url: str, model_id: str) -> Optional[List[str]]:
+ """List a running audiocpp_server's voices for MODEL_ID.
+
+ Queries ``GET /v1/audio/voices?model=<id>`` — the endpoint the converter
+ validates ``--voice`` against — and returns its voice-name list, or None
+ when the server cannot be queried. Lets the hub offer a remote server's
+ voices without reading its configuration locally.
+ """
+ query = urllib.parse.urlencode({"model": model_id})
+ try:
+ with urllib.request.urlopen(
+ f"{api_url.rstrip('/')}/v1/audio/voices?{query}",
+ timeout=10) as response:
+ payload = json.loads(response.read().decode("utf-8"))
+ except (OSError, ValueError):
+ return None
+ voices = payload.get("voices") if isinstance(payload, dict) else None
+ if not isinstance(voices, list):
+ return None
+ return [str(voice) for voice in voices]
+
+
diff --git a/app/backends/audiocpp/status.py b/app/backends/audiocpp/status.py
new file mode 100644
index 0000000..9ccf8cc
--- /dev/null
+++ b/app/backends/audiocpp/status.py
@@ -0,0 +1,90 @@
+"""detect() — the BackendStatus report for the hub's backend menu."""
+
+from typing import List, Tuple
+
+from backends import (BackendStatus, ServerSpec, format_launch_hint,
+ probe, servers)
+from converter import config
+from . import build as _build
+from . import models as _models
+
+def detect() -> BackendStatus:
+ """Detect how far audio.cpp is set up, plus the command to start it."""
+ checkout = _build.find_local_checkout()
+ details: List[str] = []
+ launch = ""
+ if checkout is None:
+ # No local checkout: only a remote server can make this usable.
+ remote = _detect_remote()
+ return BackendStatus("audiocpp", "audio.cpp", installed=False,
+ configured=False, running=remote[0],
+ remote=remote[0], remote_urls=remote[1],
+ details=["not cloned — run setup to clone "
+ "./app/audio.cpp"])
+ details.append(f"checkout: {checkout}")
+ binary = _build.find_audiocpp_server_bin(checkout)
+ built = binary is not None
+ if built:
+ details.append(f"built: {binary}")
+ else:
+ details.append("not built — run setup to build audiocpp_server")
+ server_json = checkout / "server.json"
+ configured = server_json.exists()
+ specs: List[ServerSpec] = []
+ missing = _models.missing_model_entries(server_json) if configured else []
+ if configured:
+ details.append(f"config: {server_json}")
+ if missing:
+ # The config references model files that are not on disk; a
+ # conversion would fail at model-load time, so say so now.
+ details.extend(_models.model_install_hints(checkout, missing))
+ if built:
+ # Spawned from the checkout: audiocpp_server discovers
+ # model_specs/<family>.json relative to its working directory.
+ specs = [ServerSpec(
+ "audiocpp", config.AUDIOCPP_API_URL,
+ [str(binary), "--config", str(server_json)],
+ cwd=checkout, identity=probe.IDENTITY_AUDIOCPP)]
+ else:
+ launch = (f"cd {checkout} && ./build/<platform>-<backend>-release"
+ f"/bin/audiocpp_server --config {server_json}")
+ else:
+ details.append("no server.json — run setup to configure models")
+ if specs:
+ launch = format_launch_hint(specs)
+ managed = servers.manages(specs)
+ remote_running, remote_urls = _detect_remote(managed)
+ # A more specific "part-way set up" label than unavailable/installed:
+ # cloned but never built, or built but not configured.
+ partial = ""
+ if not built:
+ partial = "downloaded (not built)"
+ elif not configured:
+ partial = "built (not configured)"
+ return BackendStatus("audiocpp", "audio.cpp", installed=built,
+ configured=configured,
+ running=managed or remote_running,
+ details=details, launch_hint=launch,
+ servers=specs, managed=managed,
+ remote=remote_running, remote_urls=remote_urls,
+ models_missing=bool(missing), partial=partial)
+
+
+def _detect_remote(managed: bool = False) -> Tuple[bool, dict]:
+ """Detect an externally-run audiocpp_server at the remote URL.
+
+ Returns ``(running, {spec_name: url})``. The remote URL is probed only
+ when configured (non-empty); a server answering there is ignored when it
+ is this tool's own managed server (remote URL == local URL and our pid is
+ still alive) — that instance is already reported as "[local]".
+ """
+ url = (config.AUDIOCPP_REMOTE_URL or "").strip()
+ if not url:
+ return False, {}
+ if managed and probe.same_endpoint(url, config.AUDIOCPP_API_URL):
+ return False, {}
+ if probe.identify_server(url) == probe.IDENTITY_AUDIOCPP:
+ return True, {"audiocpp": url}
+ return False, {}
+
+
diff --git a/app/backends/audiocpp/voices.py b/app/backends/audiocpp/voices.py
new file mode 100644
index 0000000..2f0fdd7
--- /dev/null
+++ b/app/backends/audiocpp/voices.py
@@ -0,0 +1,146 @@
+"""Reference-.wav transcription planning and execution."""
+
+import argparse
+from pathlib import Path
+from typing import Callable, Dict, List, Optional, Tuple
+
+from backends.common import (PROMPT_TEXT_FILENAME, find_wav_files,
+ read_prompt_text)
+from converter.clients import (transcribe_reference_audio,
+ whisper_backend_available)
+
+def transcribe_wav_dir(wav_files: list, whisper_model: str,
+ cancel=None) -> Dict[str, str]:
+ """Transcribe each wav file and return a mapping of stem -> transcript.
+
+ CANCEL (a ``threading.Event``) is checked between files so the in-TUI
+ task view can stop a long transcription early.
+ """
+ transcripts: Dict[str, str] = {}
+ for wav_file in wav_files:
+ if cancel is not None and cancel.is_set():
+ print("[INFO] Transcription cancelled")
+ break
+ name = wav_file.stem
+ print(f"[INFO] Transcribing {wav_file.name} (voice '{name}')...")
+ text = transcribe_reference_audio(str(wav_file), model_name=whisper_model)
+ if text:
+ print(f"[OK] {name}: {text}")
+ else:
+ print(f"[WARNING] No transcript for '{name}'; cloning works best "
+ "with an accurate transcript — consider editing prompt_text "
+ "by hand before starting the server")
+ transcripts[name] = text or ""
+ return transcripts
+
+
+def print_empty_transcript_warning(transcripts: Dict[str, str]) -> None:
+ """Print a loud, final warning for voices whose transcript is empty."""
+ empty = sorted(name for name, text in transcripts.items() if not text)
+ if not empty:
+ return
+ bar = "=" * 70
+ print()
+ print(bar)
+ print("[WARNING] MANUAL TRANSCRIPTION REQUIRED")
+ print(bar)
+ listing = " - " + "\n - ".join(empty) if len(empty) > 1 else f" - {empty[0]}"
+ print(f"The following voice(s) have an EMPTY transcript in prompt_text:\n"
+ f"{listing}")
+ print("Those voices will NOT work until you add an accurate transcript.")
+ print(f"Edit {PROMPT_TEXT_FILENAME} in your voice directory and fill in the "
+ "text after '|' for each voice above.")
+ print(bar)
+
+
+def _decide_transcription(wav_files: list, existing: Dict[str, str],
+ prompt_exists: bool, force: bool,
+ confirm: Callable[[str, bool], bool]) -> dict:
+ """Decide which voices to transcribe; CONFIRM asks the plan questions.
+
+ Returns a plan dict: {"mode": "all"|"missing"|"keep", "missing":
+ [...], "existing": {...}} — "existing" carries the prompt_text
+ mapping read while deciding, so the caller can reuse it instead of
+ reading the file again.
+ """
+ mode = "all"
+ missing: List[Path] = []
+ if prompt_exists and not force:
+ missing = [wav for wav in wav_files
+ if not existing.get(wav.stem, "").strip()]
+ if not missing:
+ if confirm("All voices already transcribed in prompt_text. "
+ "Re-transcribe anyway?", False):
+ mode = "all"
+ else:
+ mode = "keep"
+ elif confirm("Existing transcription and new .wavs detected, "
+ "only transcribe new voices?", True):
+ mode = "missing"
+ else:
+ mode = "all"
+ return {"mode": mode, "missing": missing, "existing": existing}
+
+
+def _transcribe(args: argparse.Namespace, plan: Optional[dict],
+ cancel=None) -> Tuple[Dict[str, str], bool]:
+ """Transcribe the wav directory into a stem -> transcript mapping.
+
+ Returns the mapping and a flag indicating whether it should be written to
+ prompt_text (False when an existing, complete prompt_text is kept as-is).
+ PLAN is always pre-collected — by the TUI (via _decide_transcription and
+ its confirm callbacks) or by _flag_plan for a non-interactive run — so no
+ questions are asked here; a None PLAN defaults to "transcribe everything".
+ CANCEL is checked between files.
+ """
+ wav_files = find_wav_files(args.input_dir)
+ if not wav_files:
+ print(f"[WARNING] No .wav files found in {args.input_dir}; writing the "
+ "config without a voice_dir")
+ return {}, False
+
+ prompt_path = args.input_dir / PROMPT_TEXT_FILENAME
+ existing = dict((plan or {}).get("existing") or {})
+ mode = plan["mode"] if plan else "all"
+
+ if mode == "keep":
+ print(f"[INFO] Kept existing {prompt_path}; all voices were "
+ "already transcribed, nothing new to transcribe")
+ return existing, False
+
+ if whisper_backend_available() is None:
+ print("[WARNING] Neither faster_whisper nor whisper was found, so "
+ "reference .wav files cannot be transcribed automatically and "
+ "every transcript will be empty.")
+ print(" Install whisper (or faster_whisper) in your "
+ "audiobook environment to transcribe automatically; otherwise "
+ "transcripts must be added by hand (see the warning at the end).")
+
+ if plan["mode"] == "missing":
+ new_transcripts = transcribe_wav_dir(plan["missing"], args.whisper_model,
+ cancel=cancel)
+ transcripts = dict(existing)
+ transcripts.update(new_transcripts)
+ else:
+ transcripts = transcribe_wav_dir(wav_files, args.whisper_model,
+ cancel=cancel)
+ return transcripts, True
+
+
+def _flag_plan(wav_files: list, prompt_path: Path, force: bool) -> dict:
+ """Build a transcription plan for a non-interactive (flag-only) run.
+
+ With --force everything is re-transcribed; otherwise an existing
+ prompt_text is reused and only voices with an empty transcript are
+ re-transcribed, mirroring what the TUI confirms interactively.
+ """
+ if prompt_path.exists() and not force:
+ existing = read_prompt_text(prompt_path)
+ missing = [wav for wav in wav_files
+ if not existing.get(wav.stem, "").strip()]
+ if not missing:
+ return {"mode": "keep", "missing": [], "existing": existing}
+ return {"mode": "missing", "missing": missing, "existing": existing}
+ return {"mode": "all", "missing": [], "existing": {}}
+
+
diff --git a/app/backends/audiocpp/wizard.py b/app/backends/audiocpp/wizard.py
new file mode 100644
index 0000000..dcab273
--- /dev/null
+++ b/app/backends/audiocpp/wizard.py
@@ -0,0 +1,1081 @@
+"""The audio.cpp setup wizard: TUI screens, task lanes, CLI entry points."""
+
+import argparse
+import json
+import sys
+from pathlib import Path
+from typing import Callable, Dict, List, Optional, Tuple
+
+from backends import common
+from backends.common import (
+ APP_DIR,
+ PROMPT_TEXT_FILENAME,
+ TTS_ROOT,
+ VOICES_DIR,
+ detect_wav_dir,
+ find_wav_files,
+ read_prompt_text,
+ resolve_wav_dir_arg,
+ wav_dir_info as _wav_dir_info,
+ wav_dir_preview as _wav_dir_preview,
+ write_prompt_text,
+)
+from converter import config
+from ui import taskview, tui
+from . import build as _build
+from . import configsync as _configsync
+from . import models as _models
+from . import voices as _voices
+from .catalog import (BACKENDS, DEFAULT_HOST, _backend_options,
+ build_model_entry, build_server_config, detect_backend,
+ load_model_catalog, load_server_config,
+ package_dir_options, server_config_selections)
+from .constants import (AUDIOCPP_DIR_NAME, AUDIOCPP_GIT_URL,
+ TASK_TTS, TASK_VDES)
+
+_GO_BACK = object()
+
+
+class _GoBack(Exception):
+ """Internal signal: Esc was pressed inside one of a screen's sub-prompts.
+
+ The wizard drives a stack of screens via ``tui.Wizard``. Helpers that ask
+ several questions through callbacks (the task/id pickers inside
+ ``_build_entries``, the transcription plan, the download prompt) cannot
+ themselves return the wizard's ``BACK`` sentinel, so they convert the
+ ``_GO_BACK`` value passed to each widget into this exception. The screen
+ that invoked the helper catches it and returns ``tui.Wizard.BACK``, which
+ pops back to the previous screen. Esc on the first screen aborts the
+ whole wizard.
+ """
+
+
+class _TuiError(Exception):
+ """A fatal error raised from inside the TUI wizard.
+
+ The message is reported to stderr after the terminal is restored; the
+ process exits with code 2 (matching a parser error).
+ """
+
+
+# Alias kept on this module: main()'s tty check and its tests patch it
+# here.
+from backends.setup import interactive as _interactive
+
+
+def _build_entries(family_keys: List[str], chosen: Dict[str, List[dict]],
+ catalog_by_family: Dict[str, dict],
+ task_picker: Callable[[str], str],
+ known_tasks: Optional[Dict[Tuple[str, str], str]] = None
+ ) -> Tuple[List[dict], List[str], List[Tuple[str, str]],
+ List[str], bool]:
+ """Build server.json model entries from the selected families/packages.
+
+ TASK_PICKER is called for each design package to choose vdes/tts.
+ KNOWN_TASKS maps ``(family, target_directory)`` to a previously-stored
+ task ("tts" or "vdes") so a modify run preserves how a design package
+ was hosted instead of re-asking. Each entry's server id is its package
+ ``target_directory`` (flattened to a token), so packages from the same
+ family never collide; an id that does collide (across families) is
+ auto-suffixed without prompting. Returns (model_entries, entry_ids,
+ install_guidance, design_entry_ids, include_clone).
+ """
+ model_entries: List[dict] = []
+ entry_ids: List[str] = []
+ install_guidance: List[Tuple[str, str]] = []
+ design_entry_ids: List[str] = []
+ include_clone = False
+ for family in family_keys:
+ entry = catalog_by_family[family]
+ include_clone = include_clone or entry["clone_capable"]
+ for opt in chosen[family]:
+ if opt["design"]:
+ task = known_tasks.get((family, opt["target_directory"])) \
+ if known_tasks else None
+ if task is None:
+ task = task_picker(opt["install_id"])
+ else:
+ task = TASK_TTS
+ base_id = opt["target_directory"].replace("/", "-")
+ model_id = base_id
+ if model_id in entry_ids:
+ n = 2
+ while f"{base_id}-{n}" in entry_ids:
+ n += 1
+ model_id = f"{base_id}-{n}"
+ entry_ids.append(model_id)
+ model_entries.append(build_model_entry(
+ family, model_id, f"models/{opt['target_directory']}",
+ task=task))
+ install_guidance.append((entry["display_name"], opt["install_id"]))
+ if task == TASK_VDES:
+ design_entry_ids.append(model_id)
+ return (model_entries, entry_ids, install_guidance,
+ design_entry_ids, include_clone)
+
+
+def _write_and_advise(audiocpp_dir: Path, wav_dir: Optional[Path],
+ output_path: Path, model_entries: List[dict],
+ install_guidance: List[Tuple[str, str]], host: str,
+ port: int, backend: str, lazy_load: bool,
+ transcripts: Dict[str, str], write_prompt: bool) -> None:
+ """Console phase shared by both UI modes: write files, print summary.
+
+ After a successful run the console output is the path of the written
+ server.json. The model install commands (and optional automatic
+ download) are handled separately by _install_models, called by both
+ UI modes once the user has decided whether to download.
+ """
+ voice_dir: Optional[str] = None
+ if transcripts:
+ if write_prompt:
+ prompt_path = wav_dir / PROMPT_TEXT_FILENAME
+ write_prompt_text(wav_dir, transcripts)
+ print(f"[OK] Wrote {prompt_path}")
+ voice_dir = str(wav_dir.resolve())
+
+ server_config = build_server_config(
+ host=host, port=port, backend=backend, lazy_load=lazy_load,
+ model_entries=model_entries, voice_dir=voice_dir)
+
+ with output_path.open("w", encoding="utf-8") as handle:
+ json.dump(server_config, handle, indent=2, ensure_ascii=False)
+ handle.write("\n")
+
+ count = len(model_entries)
+ print(f"Wrote {output_path.resolve()} with {count} "
+ f"{'entry' if count == 1 else 'entries'}.")
+
+
+def _wizard(stdscr, args: argparse.Namespace, parser: argparse.ArgumentParser
+ ) -> Optional[dict]:
+ """Run every TUI screen; return the collected settings, or None to abort.
+
+ The wizard is driven by ``tui.Wizard`` as a stack of screen closures:
+ each screen shows one interactive widget and returns the next screen
+ (a closure), ``Wizard.BACK`` (Esc/q pressed — pop to the previous
+ screen), or the final settings dict. Only screens that actually render
+ are pushed, so Esc always lands on the previous real screen. A step
+ whose value is already provided by a flag (``--host``, ``--port``,
+ ``--families``, ...) or does not apply (e.g. the port-sync prompt when
+ the port did not change) is folded into the ``_after_*`` guards and
+ never becomes a screen. Esc on the first screen aborts the whole
+ wizard.
+ """
+
+ s: dict = {}
+
+ def ask_confirm(question: str, default: bool) -> bool:
+ result = tui.confirm(stdscr, question, default=default,
+ cancel_value=_GO_BACK)
+ if result is _GO_BACK:
+ raise _GoBack()
+ return result
+
+ def resolve_checkout(audiocpp_dir: Path) -> None:
+ """Validate the audio.cpp checkout and populate the wizard state ``s``."""
+ audiocpp_dir = Path(audiocpp_dir).resolve()
+ try:
+ catalog = load_model_catalog(audiocpp_dir)
+ except NotADirectoryError as exc:
+ raise _TuiError(str(exc))
+ if not catalog:
+ raise _TuiError(f"No TTS model families found in "
+ f"{audiocpp_dir}/model_specs; check the "
+ "checkout is up to date")
+ catalog_by_family = {entry["family"]: entry for entry in catalog}
+ output_path = args.output if args.output is not None \
+ else audiocpp_dir / "server.json"
+ # Modify flow: an existing server.json seeds the wizard's screens
+ # instead of being overwritten from scratch (an explicit --force
+ # still starts fresh).
+ existing_config = load_server_config(output_path) \
+ if not args.force else None
+ if existing_config is not None:
+ existing_selected, existing_tasks = \
+ server_config_selections(existing_config, catalog)
+ else:
+ existing_selected, existing_tasks = {}, {}
+ s.update({
+ "audiocpp_dir": audiocpp_dir,
+ "catalog": catalog,
+ "catalog_by_family": catalog_by_family,
+ "output_path": output_path,
+ "existing_config": existing_config,
+ "existing_selected": existing_selected,
+ "existing_tasks": existing_tasks,
+ "existing_host": existing_config.get("host")
+ if existing_config else None,
+ "existing_port": existing_config.get("port")
+ if existing_config else None,
+ "existing_backend": existing_config.get("backend")
+ if existing_config else None,
+ "existing_voice_dir": existing_config.get("voice_dir")
+ if existing_config else None,
+ "detected_backend": detect_backend(audiocpp_dir),
+ })
+
+ def _families_from_flag() -> None:
+ requested = [f.strip() for f in args.families.split(",") if f.strip()]
+ unknown = [f for f in requested if f not in s["catalog_by_family"]]
+ if unknown:
+ raise _TuiError(
+ f"Unknown family in --families: {', '.join(unknown)}. "
+ f"Available: {', '.join(s['catalog_by_family'])}")
+ chosen: Dict[str, List[dict]] = {}
+ family_keys: List[str] = []
+ for family in requested:
+ if family not in family_keys:
+ family_keys.append(family)
+ chosen[family] = [opt for opt in package_dir_options(
+ s["catalog_by_family"][family]) if opt["recommended"]]
+ s["chosen"] = chosen
+ s["family_keys"] = family_keys
+
+ def _compute_entries() -> None:
+ # Design task menu. Esc raises _GoBack, which the caller turns into
+ # Wizard.BACK (the design prompts are grouped: Esc returns to the
+ # families tree).
+ def task_picker(install_id: str) -> str:
+ result = tui.menu(
+ stdscr,
+ f"How should the '{install_id}' package be hosted?",
+ [
+ ("design (vdes) - describe the voice with "
+ "--instructions", TASK_VDES),
+ ("tts - normal synthesis", TASK_TTS),
+ ], default_index=0, back_value=_GO_BACK)
+ if result is _GO_BACK:
+ raise _GoBack()
+ return result
+
+ model_entries, entry_ids, install_guidance, \
+ design_entry_ids, include_clone = _build_entries(
+ s["family_keys"], s["chosen"], s["catalog_by_family"],
+ task_picker, known_tasks=s["existing_tasks"])
+ s.update({
+ "model_entries": model_entries,
+ "entry_ids": entry_ids,
+ "install_guidance": install_guidance,
+ "design_entry_ids": design_entry_ids,
+ "include_clone": include_clone,
+ })
+
+ def _finalize() -> dict:
+ return {
+ "audiocpp_dir": s["audiocpp_dir"],
+ "catalog": s["catalog"],
+ "catalog_by_family": s["catalog_by_family"],
+ "output_path": s["output_path"],
+ "family_keys": s["family_keys"],
+ "chosen": s["chosen"],
+ "model_entries": s["model_entries"],
+ "entry_ids": s["entry_ids"],
+ "install_guidance": s["install_guidance"],
+ "design_entry_ids": s["design_entry_ids"],
+ "include_clone": s["include_clone"],
+ "host": s["host"],
+ "port": s["port"],
+ "backend": s["backend"],
+ "build": s["build"],
+ "lazy_load": s["lazy_load"],
+ "sync_port": s["sync_port"],
+ "sync_model_ids": s["sync_model_ids"],
+ "wav_dir": s["wav_dir"],
+ "plan": s["plan"],
+ "download": s["download"],
+ "delete_unused": s["delete_unused"],
+ "unused_entries": s["unused_entries"],
+ }
+
+ def screen_families():
+ """Pick TTS model families and packages (the modify tree)."""
+ tree_families = _models._build_tree_families(s["catalog"])
+ # Modify flow: pre-check the models an existing server.json hosts,
+ # so the tree opens as a "modify" list rather than a fresh one.
+ checked_set = set()
+ for family, dirs in s["existing_selected"].items():
+ if family not in s["catalog_by_family"]:
+ continue
+ family_index = s["catalog"].index(s["catalog_by_family"][family])
+ valid_dirs = {opt["target_directory"]
+ for opt in package_dir_options(
+ s["catalog_by_family"][family])}
+ for target in dirs:
+ if target in valid_dirs:
+ checked_set.add((family_index, target))
+ picked = tui.checkbox_tree(
+ stdscr, "Select TTS model families to host",
+ tree_families, expand_all=args.all_packages,
+ back_value=_GO_BACK, checked=checked_set)
+ if picked is _GO_BACK:
+ return tui.Wizard.BACK
+ chosen: Dict[str, List[dict]] = {}
+ family_keys: List[str] = []
+ for family_index, option_key in picked:
+ family = s["catalog"][family_index]["family"]
+ if family not in chosen:
+ chosen[family] = []
+ family_keys.append(family)
+ chosen[family].append(option_key)
+ for family in list(chosen):
+ keyed = {opt["target_directory"]: opt
+ for opt in package_dir_options(
+ s["catalog_by_family"][family])}
+ chosen[family] = [keyed[key] for key in chosen[family]]
+ s["chosen"] = chosen
+ s["family_keys"] = family_keys
+ return screen_host
+
+ def _after_families():
+ if args.families is not None:
+ _families_from_flag()
+ return screen_host
+ return screen_families
+
+ def screen_host():
+ """Build the model entries, then ask the bind host.
+
+ The task/id pickers (when any) run here too and are grouped with
+ this screen: Esc on one of them (or on the host field) returns to
+ the families tree.
+ """
+ try:
+ _compute_entries()
+ except _GoBack:
+ return tui.Wizard.BACK
+ if args.host is not None:
+ s["host"] = args.host
+ return _after_host()
+ host = tui.line_edit(
+ stdscr, "Bind host",
+ s["existing_host"] if isinstance(s["existing_host"], str)
+ else DEFAULT_HOST,
+ help_lines=["The IP address audiocpp will be hosted on",
+ "127.0.0.1 (this machine) is probably "
+ "correct"], back_value=_GO_BACK)
+ if host is _GO_BACK:
+ return tui.Wizard.BACK
+ s["host"] = host
+ return _after_host()
+
+ def _after_host():
+ if args.port is None:
+ return screen_port
+ s["port"] = args.port
+ return _after_port()
+
+ def screen_port():
+ port_text = tui.line_edit(
+ stdscr, "Port",
+ str(s["existing_port"]) if isinstance(s["existing_port"], int)
+ else str(_configsync.config_port()),
+ validate=lambda s: None if (s.isdigit()
+ and 1 <= int(s) <= 65535)
+ else "Enter a port number between 1 and 65535",
+ help_lines=["The port audiocpp will be hosted on"],
+ back_value=_GO_BACK)
+ if port_text is _GO_BACK:
+ return tui.Wizard.BACK
+ s["port"] = int(port_text)
+ return _after_port()
+
+ def _after_port():
+ s["sync_port"] = None
+ if s["port"] != _configsync.config_port():
+ return screen_sync_port
+ return _after_sync()
+
+ def screen_sync_port():
+ sync_port = tui.confirm(
+ stdscr, "Update AUDIOCPP_API_URL in app/converter/config.py "
+ f"to port {s['port']} so audiobook.py talks to this server",
+ default=True, cancel_value=_GO_BACK)
+ if sync_port is _GO_BACK:
+ return tui.Wizard.BACK
+ s["sync_port"] = sync_port
+ return _after_sync()
+
+ def _after_sync():
+ if args.build_backend:
+ s["backend"] = args.build_backend
+ s["build"] = s["detected_backend"] is None
+ return _after_backend()
+ if args.backend:
+ s["backend"] = args.backend
+ s["build"] = False
+ return _after_backend()
+ if s["detected_backend"] is not None:
+ # Already built: use the detected backend, no menu, no build.
+ s["backend"] = s["detected_backend"]
+ s["build"] = False
+ return _after_backend()
+ # Not built for any backend yet: always ask which backend the server
+ # should use and offer to build it — even on a modify run, so a user
+ # who declined the build the first time is never stranded without a
+ # way to build from the TUI.
+ return screen_backend
+
+ def screen_backend():
+ # Pre-select the backend an existing server.json records (modify
+ # flow), so re-running setup lands on the previous choice.
+ backend_options, backend_default = _backend_options(None)
+ if s["existing_backend"] in BACKENDS:
+ backend_default = next(
+ (index for index, (_label, value) in enumerate(backend_options)
+ if value == s["existing_backend"]), backend_default)
+ backend = tui.menu(
+ stdscr, "Which inference backend should audiocpp_server "
+ "use?", backend_options,
+ default_index=backend_default, back_value=_GO_BACK)
+ if backend is _GO_BACK:
+ return tui.Wizard.BACK
+ s["backend"] = backend
+ if _build.built_server_binary(s["audiocpp_dir"], backend) is not None:
+ # A checkout with builds for several backends: this one is
+ # already built, so there is nothing to build.
+ s["build"] = False
+ return _after_backend()
+ return screen_build
+
+ def screen_build():
+ # Not built for the chosen backend yet: offer to build it now. The
+ # build itself runs in the TUI task view (or the console tail for
+ # CLI runs) after the wizard.
+ build = tui.confirm(
+ stdscr, f"audiocpp_server is not built for {s['backend']}. "
+ f"Build it now (runs scripts/build_*)?",
+ default=True, cancel_value=_GO_BACK)
+ if build is _GO_BACK:
+ return tui.Wizard.BACK
+ s["build"] = build
+ return _after_backend()
+
+ def _after_backend():
+ s["lazy_load"] = True
+ return _after_lazy()
+
+ def _after_lazy():
+ if args.input_dir is not None:
+ s["wav_dir"] = args.input_dir
+ return _after_wav()
+ if s["include_clone"]:
+ return screen_wav
+ s["wav_dir"] = None
+ return _after_wav()
+
+ def screen_wav():
+ wav_start = detect_wav_dir(s["audiocpp_dir"], TTS_ROOT)
+ # Modify flow: an existing voice_dir seeds the browser so the user
+ # can accept it on Enter instead of re-navigating.
+ if isinstance(s["existing_voice_dir"], str) and s["existing_voice_dir"]:
+ wav_start = Path(s["existing_voice_dir"])
+ wav_dir = tui.browse_directory(
+ stdscr, "Select the directory with your .wav voices",
+ info=_wav_dir_info, preview=_wav_dir_preview,
+ start=wav_start if wav_start is not None else VOICES_DIR,
+ back_value=_GO_BACK)
+ if wav_dir is _GO_BACK:
+ return tui.Wizard.BACK
+ s["wav_dir"] = wav_dir
+ return _after_wav()
+
+ def _after_wav():
+ s["plan"] = None
+ if s["include_clone"] and s["wav_dir"] is not None:
+ wav_files = find_wav_files(s["wav_dir"])
+ if wav_files:
+ prompt_path = s["wav_dir"] / PROMPT_TEXT_FILENAME
+ if prompt_path.exists() and not args.force:
+ return screen_transcription
+ existing = read_prompt_text(prompt_path) if (
+ prompt_path.exists() and not args.force) else {}
+ s["plan"] = _voices._decide_transcription(
+ wav_files, existing, prompt_path.exists(),
+ args.force, ask_confirm)
+ return _after_transcription()
+
+ def screen_transcription():
+ # Transcription plan (questions only; transcription runs after).
+ wav_files = find_wav_files(s["wav_dir"])
+ prompt_path = s["wav_dir"] / PROMPT_TEXT_FILENAME
+ existing = read_prompt_text(prompt_path) if (
+ prompt_path.exists() and not args.force) else {}
+ try:
+ s["plan"] = _voices._decide_transcription(
+ wav_files, existing, prompt_path.exists(),
+ args.force, ask_confirm)
+ except _GoBack:
+ return tui.Wizard.BACK
+ return _after_transcription()
+
+ def _after_transcription():
+ s["sync_model_ids"] = None
+ if len(s["entry_ids"]) == 1 and not (
+ config.AUDIOCPP_MODEL_ID == s["entry_ids"][0]
+ and config.AUDIOCPP_CLONE_MODEL_ID == s["entry_ids"][0]):
+ return screen_model_sync
+ return _after_model_sync()
+
+ def screen_model_sync():
+ sync_model_ids = tui.confirm(
+ stdscr, "Update AUDIOCPP_MODEL_ID and "
+ "AUDIOCPP_CLONE_MODEL_ID in app/converter/config.py to "
+ f"'{s['entry_ids'][0]}' so audiobook.py uses this model",
+ default=True, cancel_value=_GO_BACK)
+ if sync_model_ids is _GO_BACK:
+ return tui.Wizard.BACK
+ s["sync_model_ids"] = sync_model_ids
+ return _after_model_sync()
+
+ def _after_model_sync():
+ new_paths = {entry["path"] for entry in s["model_entries"]}
+ s["unused_entries"] = _models.unused_installed_entries(
+ s["output_path"], new_paths) \
+ if s["existing_config"] is not None else []
+ s["delete_unused"] = False
+ if s["unused_entries"]:
+ return screen_delete_unused
+ return _after_delete()
+
+ def screen_delete_unused():
+ delete_unused = tui.confirm(
+ stdscr, "Delete unused models?", default=False,
+ cancel_value=_GO_BACK)
+ if delete_unused is _GO_BACK:
+ return tui.Wizard.BACK
+ s["delete_unused"] = delete_unused
+ return _after_delete()
+
+ def _after_delete():
+ manager = s["audiocpp_dir"] / "tools" / "model_manager_v2.py"
+ if manager.is_file():
+ return screen_download
+ s["download"] = False
+ return _finalize()
+
+ def screen_download():
+ # Automatic model download (or print the install commands).
+ try:
+ s["download"] = _models._decide_download(
+ s["audiocpp_dir"], s["model_entries"], ask_confirm)
+ except _GoBack:
+ return tui.Wizard.BACK
+ return _finalize()
+
+ # First screen: resolve the checkout directly when it already exists
+ # (the modify flow), so the wizard starts on a real screen. When no
+ # checkout exists, clone it into ./app/audio.cpp (streaming inside the
+ # TUI task view, not by dropping to the console) without asking, then
+ # continue the same way.
+ audiocpp_dir = _build.find_local_checkout()
+ if audiocpp_dir is None:
+ target = APP_DIR / AUDIOCPP_DIR_NAME
+ rc = taskview.run_steps(stdscr, "Clone audio.cpp", [
+ taskview.TaskStep(
+ f"Cloning audio.cpp into {target}",
+ lambda emit, cancel: common.git_clone(
+ AUDIOCPP_GIT_URL, target, emit=emit, cancel=cancel)),
+ taskview.TaskStep(
+ "Apply ggml build patches",
+ lambda emit, cancel: _build.apply_ggml_patches(
+ target, emit=emit, cancel=cancel)),
+ ])
+ if rc == 130:
+ # Cancelled from the task view: abort the wizard quietly.
+ return None
+ if rc != 0:
+ raise _TuiError(
+ f"audio.cpp setup step failed (exit {rc}). Clone "
+ f"audio.cpp manually: git clone "
+ f"{AUDIOCPP_GIT_URL} {target}, then re-run")
+ audiocpp_dir = target
+ resolve_checkout(audiocpp_dir)
+ first = _after_families()
+ return tui.Wizard().run(first)
+
+
+def _execute_lanes(settings: dict,
+ args: argparse.Namespace) -> List[taskview.TaskLane]:
+ """Build the ordered setup steps for the in-TUI task view, per lane.
+
+ The same work ``_execute`` runs on the console, split into two lanes so
+ the view can run the build in one pane while configuring and downloading
+ models in the other (both progress bars visible at once). The build lane
+ exists only when ``settings["build"]`` is set; the models lane always
+ exists (transcribe → write server.json → download/print commands).
+ Shared results (the transcription mapping) travel through a small closure
+ dict scoped to the models lane. Each step's ``work(emit, cancel)``
+ returns its exit code; subprocess steps stream through EMIT and abort on
+ CANCEL, while print()-based steps are captured by the view's stdout
+ routing.
+ """
+ audiocpp_dir = settings["audiocpp_dir"]
+ state: dict = {}
+ build = settings.get("build")
+ lanes: List[taskview.TaskLane] = []
+
+ if build:
+ def build_step(emit, cancel):
+ rc = _build.build_audiocpp(audiocpp_dir, settings["backend"],
+ emit=emit, cancel=cancel)
+ if rc != 0:
+ print(f"[WARNING] build exited with code {rc}; the server.json "
+ "was still written — build audiocpp_server manually "
+ "before starting it")
+ else:
+ print("[OK] build complete")
+ return rc
+ lanes.append(taskview.TaskLane(
+ "Build",
+ [taskview.TaskStep(
+ f"Build audiocpp_server ({settings['backend']})",
+ build_step)]))
+
+ def transcribe(emit, cancel):
+ args.input_dir = settings["wav_dir"]
+ if settings["include_clone"] and args.input_dir is not None:
+ transcripts, write_prompt = _voices._transcribe(
+ args, plan=settings["plan"], cancel=cancel)
+ elif args.input_dir is not None:
+ print(f"[WARNING] Ignoring {args.input_dir}: no clone-capable "
+ "family selected, so voice presets are not used")
+ transcripts, write_prompt = {}, False
+ else:
+ transcripts, write_prompt = {}, False
+ state["transcripts"] = transcripts
+ state["write_prompt"] = write_prompt
+ return 0
+
+ def write(emit, cancel):
+ # Port sync (applied now that the terminal is back).
+ if settings["sync_port"] is True:
+ _configsync._apply_port_sync(settings["port"], True)
+ elif settings["sync_port"] is False:
+ _configsync._apply_port_sync(settings["port"], False)
+
+ _write_and_advise(
+ audiocpp_dir, settings["wav_dir"], settings["output_path"],
+ settings["model_entries"], settings["install_guidance"],
+ settings["host"], settings["port"], settings["backend"],
+ settings["lazy_load"], state["transcripts"], state["write_prompt"])
+
+ # Delete-unused cleanup (modify flow): remove the already-downloaded
+ # models the new selection dropped. The regenerated server.json
+ # already only lists the kept entries.
+ if settings.get("delete_unused"):
+ removed = _models.delete_model_files(settings["output_path"],
+ settings["unused_entries"])
+ print(f"[OK] Deleted {removed} unused model "
+ f"{'entry' if removed == 1 else 'entries'} from disk.")
+
+ if len(settings["entry_ids"]) == 1:
+ _configsync._offer_config_model_id_sync(settings["entry_ids"][0],
+ settings["sync_model_ids"])
+ _voices.print_empty_transcript_warning(state["transcripts"])
+ return 0
+
+ def install(emit, cancel):
+ _models._install_models(audiocpp_dir, settings["install_guidance"],
+ settings["download"], emit=emit, cancel=cancel)
+ _build._print_launch_hint(audiocpp_dir, settings["output_path"])
+ return 0
+ install_title = "Download models" if settings.get("download") \
+ else "Print model install commands"
+
+ lanes.append(taskview.TaskLane(
+ "Configure & download",
+ [taskview.TaskStep("Transcribe reference voices", transcribe),
+ taskview.TaskStep("Write server.json & sync config", write),
+ taskview.TaskStep(install_title, install)]))
+
+ return lanes
+
+
+def _execute_steps(settings: dict,
+ args: argparse.Namespace) -> List[taskview.TaskStep]:
+ """The ordered setup steps for the sequential console path.
+
+ The lanes ``_execute_lanes`` builds, flattened into one ordered list
+ (build first, then transcribe → write → download), so the console tail
+ is byte-identical to the pre-lanes behavior.
+ """
+ steps: List[taskview.TaskStep] = []
+ for lane in _execute_lanes(settings, args):
+ steps.extend(lane.steps)
+ return steps
+
+
+def _execute(settings: dict, args: argparse.Namespace) -> int:
+ """Shared console tail: build, sync, transcribe, write, install, advise.
+
+ Runs after the TUI wizard returns (or after _collect_from_flags for a
+ non-interactive run): the terminal is plain, so subprocess output and
+ transcription progress appear normally. The same work as
+ ``_execute_steps``, run with no emit (console streaming).
+ """
+ return taskview.run_steps_inline(_execute_steps(settings, args))
+
+
+def setup_screen(stdscr) -> int:
+ """Run the setup wizard on an existing curses screen (the hub's).
+
+ The hub drives this as one screen of its own ``tui.Wizard`` stack, so
+ Esc on the wizard's first screen simply returns here and the hub pops
+ back to the menu that launched it. The setup tail (build, transcribe,
+ write, download) runs inside the TUI task view on this same screen, so
+ the hub's curses session stays intact and the user sees per-step status
+ and progress instead of being dropped to the console. On a fresh install
+ the build and the model setup run as two parallel lanes (a split view),
+ so cloning → configuring → building+downloading is one continuous,
+ one-click flow; the individual "Build" and "Download Missing Models" hub
+ actions remain only as fallbacks when something fails or is interrupted.
+ Returns 0 on completion, 1 when the user aborted.
+ """
+ parser = build_parser()
+ args = parser.parse_args([])
+ settings = _wizard(stdscr, args, parser)
+ if settings is None:
+ return 1
+ return taskview.run_lanes(stdscr, "Setting up audio.cpp",
+ _execute_lanes(settings, args))
+
+
+def build_screen(stdscr) -> int:
+ """Build audiocpp_server from the hub when the checkout has no binary.
+
+ Asks which backend to build for (pre-selecting the backend an existing
+ server.json records, else cuda), runs the build inside the TUI task view
+ — alongside a download of any missing models when server.json is already
+ configured and those models map to an install command (the split view),
+ or just the build otherwise — then updates server.json's ``backend``
+ field to match. Returns 0 on success, non-zero when the user backed out,
+ cancelled, or the build failed. This is the hub's "Build audio.cpp
+ server" action, so a checkout that was cloned but never built is always
+ buildable from the TUI; the standalone "Download Missing Models" action
+ stays as the fallback when the download fails or is interrupted.
+ """
+ checkout = _build.find_local_checkout()
+ if checkout is None:
+ tui.flash(stdscr, "No audio.cpp checkout found — install audio.cpp "
+ "first.", "err")
+ return 1
+ if _build.find_audiocpp_server_bin(checkout) is not None:
+ tui.flash(stdscr, "audiocpp_server is already built.", "ok")
+ return 0
+ server_config = load_server_config(checkout / "server.json") or {}
+ recorded = server_config.get("backend")
+ options, default = _backend_options(None)
+ if recorded in BACKENDS:
+ default = next((i for i, (_label, value) in enumerate(options)
+ if value == recorded), default)
+ backend = tui.menu(
+ stdscr, "Which inference backend should audiocpp_server be built "
+ "for?", options, default_index=default, back_value=_GO_BACK)
+ if backend is _GO_BACK:
+ return 1
+
+ def build_step(emit, cancel):
+ return _build.build_audiocpp(checkout, backend, emit=emit, cancel=cancel)
+
+ lanes = [taskview.TaskLane(
+ "Build", [taskview.TaskStep(
+ f"Build audiocpp_server ({backend})", build_step)])]
+
+ # Missing models this build can also fetch, so a configured backend that
+ # lost its binary is restored to "installed" in one step.
+ server_json = checkout / "server.json"
+ missing = _models.missing_model_entries(server_json) if server_json.exists() else []
+ guidance = _models.missing_model_install_guidance(checkout, missing) \
+ if missing else []
+
+ if guidance:
+ def download_step(emit, cancel):
+ _models.install_models(checkout, guidance, emit=emit, cancel=cancel)
+ return 0
+ lanes.append(taskview.TaskLane(
+ "Download models",
+ [taskview.TaskStep("Download missing models", download_step)]))
+
+ title = "Build & download models" if len(lanes) == 2 \
+ else "Build audiocpp_server"
+ rc = taskview.run_lanes(stdscr, title, lanes)
+ if rc != 0:
+ return rc
+ if _configsync.update_server_backend(backend):
+ tui.flash(stdscr, f"audiocpp_server built for {backend}.", "ok")
+ else:
+ tui.flash(stdscr, f"audiocpp_server built for {backend}. (Could not "
+ "update server.json's backend field — reconfigure audio.cpp "
+ "if it was already configured.)", "warn")
+ # Models that can't be mapped to an install command still need hand
+ # installation; say so now rather than leaving the user in the dark.
+ if missing and not guidance:
+ tui.flash(stdscr, _models.hand_install_guidance(checkout, missing), "err")
+ return 0
+
+
+def run_tui(args: Optional[argparse.Namespace] = None,
+ parser: Optional[argparse.ArgumentParser] = None) -> int:
+ """Run the audio.cpp setup wizard end-to-end.
+
+ With no ARGS (the hub's call) a default namespace is built so the full
+ wizard runs. Called from ``main`` after argparse when the terminal is
+ interactive. Returns the process exit code.
+ """
+ import curses
+ if args is None:
+ parser = build_parser()
+ args = parser.parse_args([])
+ if args.input_dir is not None and not args.input_dir.is_dir():
+ print(f"[ERROR] --wavs not found: {args.input_dir}",
+ file=sys.stderr)
+ return 2
+ try:
+ settings = curses.wrapper(_wizard, args, parser)
+ except _TuiError as exc:
+ print(f"[ERROR] {exc}", file=sys.stderr)
+ return 2
+ except tui.WizardCancelled:
+ print("\n[INFO] Cancelled; nothing was written")
+ return 1
+ try:
+ curses.curs_set(1) # restore the text cursor hidden by the TUI
+ except curses.error:
+ pass
+ if settings is None:
+ print("[INFO] Aborted; existing server.json kept")
+ return 1
+ return _execute(settings, args)
+
+
+def _collect_from_flags(args: argparse.Namespace,
+ parser: argparse.ArgumentParser) -> Optional[dict]:
+ """Build the settings dict from flags for a non-interactive run.
+
+ Every required value must come from a flag (there are no prompts in a
+ non-interactive run); a missing one is a hard ``parser.error``. Returns
+ the settings dict, or None when the user declined an overwrite (the
+ default-location fallback then also exists).
+ """
+ # Checkout: ./app/audio.cpp, else --clone clones one there.
+ audiocpp_dir = _build.find_local_checkout()
+ if audiocpp_dir is None and args.clone:
+ target = APP_DIR / AUDIOCPP_DIR_NAME
+ rc = common.git_clone(AUDIOCPP_GIT_URL, target)
+ if rc != 0:
+ parser.error(f"git clone failed (exit {rc}); clone audio.cpp "
+ f"manually: git clone {AUDIOCPP_GIT_URL} {target}")
+ patch_rc = _build.apply_ggml_patches(target)
+ if patch_rc != 0:
+ parser.error(
+ f"ggml build patches could not be applied to {target} "
+ f"(exit {patch_rc}); see messages above. The audio.cpp "
+ f"fork's vendored ggml may have changed — re-evaluate "
+ f"app/backends/patches/.")
+ audiocpp_dir = target
+ if audiocpp_dir is None:
+ parser.error(
+ "An audio.cpp checkout is required. Pass --clone to clone "
+ "app/audio.cpp, or run without flags for the TUI wizard.")
+ try:
+ catalog = load_model_catalog(audiocpp_dir)
+ except NotADirectoryError as exc:
+ parser.error(str(exc))
+ if not catalog:
+ parser.error(
+ f"No TTS model families found in {audiocpp_dir}/model_specs; "
+ "check the checkout is up to date")
+ catalog_by_family = {entry["family"]: entry for entry in catalog}
+
+ # Families: required from --families in a non-interactive run.
+ if args.families is None:
+ parser.error("--families is required in a non-interactive run (or run "
+ "without flags for the TUI wizard)")
+ requested = [f.strip() for f in args.families.split(",") if f.strip()]
+ unknown = [f for f in requested if f not in catalog_by_family]
+ if unknown:
+ parser.error(
+ f"Unknown family in --families: {', '.join(unknown)}. "
+ f"Available: {', '.join(catalog_by_family)}")
+ family_keys: List[str] = []
+ for fam in requested:
+ if fam not in family_keys:
+ family_keys.append(fam)
+
+ chosen: Dict[str, List[dict]] = {}
+ for family in family_keys:
+ opts = package_dir_options(catalog_by_family[family])
+ if args.all_packages:
+ chosen[family] = opts
+ else:
+ chosen[family] = [opt for opt in opts if opt["recommended"]]
+
+ # Non-interactive picker: design packages default to vdes.
+ def task_picker(install_id: str) -> str:
+ return TASK_VDES
+
+ model_entries, entry_ids, install_guidance, design_entry_ids, include_clone = \
+ _build_entries(family_keys, chosen, catalog_by_family,
+ task_picker)
+
+ # Server settings.
+ host = args.host or DEFAULT_HOST
+ detected_backend = detect_backend(audiocpp_dir)
+ if args.build_backend:
+ backend = args.build_backend
+ build = detected_backend is None
+ elif args.backend:
+ backend = args.backend
+ build = False
+ elif detected_backend is not None:
+ backend = detected_backend
+ build = False
+ else:
+ backend = "cuda"
+ build = False
+ port = args.port if args.port is not None else _configsync.config_port()
+ lazy_load = True
+
+ # Output path / overwrite (decline falls back to cwd, then aborts).
+ output_path = args.output if args.output is not None \
+ else audiocpp_dir / "server.json"
+ if output_path.exists() and not args.force:
+ if args.output is None:
+ output_path = Path.cwd() / "server.json"
+ if output_path.exists() and not args.force:
+ print("[INFO] Aborted; existing server.json kept")
+ return None
+ else:
+ print("[INFO] Aborted; existing server.json kept")
+ return None
+
+ # Config sync decisions (auto-apply unless explicitly declined).
+ sync_port: Optional[bool] = None
+ if port != _configsync.config_port():
+ sync_port = not args.no_sync_port
+ sync_model_ids: Optional[bool] = None
+ if len(entry_ids) == 1 and not (
+ config.AUDIOCPP_MODEL_ID == entry_ids[0]
+ and config.AUDIOCPP_CLONE_MODEL_ID == entry_ids[0]):
+ sync_model_ids = not args.no_sync_model_ids
+
+ # Wav dir + transcription plan (defaults to the project's voices/ dir).
+ wav_dir = args.input_dir if args.input_dir is not None else VOICES_DIR
+ plan: Optional[dict] = None
+ if include_clone and wav_dir is not None:
+ wav_files = find_wav_files(wav_dir)
+ if wav_files:
+ prompt_path = wav_dir / PROMPT_TEXT_FILENAME
+ plan = _voices._flag_plan(wav_files, prompt_path, args.force)
+
+ return {
+ "audiocpp_dir": audiocpp_dir,
+ "catalog": catalog,
+ "catalog_by_family": catalog_by_family,
+ "output_path": output_path,
+ "family_keys": family_keys,
+ "chosen": chosen,
+ "model_entries": model_entries,
+ "entry_ids": entry_ids,
+ "install_guidance": install_guidance,
+ "design_entry_ids": design_entry_ids,
+ "include_clone": include_clone,
+ "host": host,
+ "port": port,
+ "backend": backend,
+ "build": build,
+ "lazy_load": lazy_load,
+ "sync_port": sync_port,
+ "sync_model_ids": sync_model_ids,
+ "wav_dir": wav_dir,
+ "plan": plan,
+ "download": args.download,
+ }
+
+
+def build_parser() -> argparse.ArgumentParser:
+ """The audio.cpp setup CLI (also used to build a default namespace)."""
+ parser = argparse.ArgumentParser(
+ description="Set up the audio.cpp TTS backend: clone/build, pick "
+ "models, write server.json, and sync app/converter/config.py.")
+ parser.add_argument("--wavs", type=resolve_wav_dir_arg, default=None,
+ dest="input_dir", metavar="WAV_DIR",
+ help="Directory with .wav reference files to publish as "
+ "a server-level voice_dir cloning library "
+ f"(default: {VOICES_DIR}; asked for when omitted "
+ "in the TUI)")
+ parser.add_argument("--output", type=Path, default=None,
+ help="Output path for server.json (default: "
+ "server.json inside the audio.cpp checkout; an "
+ "existing file is overwritten only with --force "
+ "or a TUI confirm)")
+ parser.add_argument("--clone", action="store_true",
+ help="Non-interactive: clone audio.cpp into "
+ "./app/audio.cpp when no checkout is found")
+ parser.add_argument("--families", type=str, default=None,
+ help="Comma-separated model families to host, as named "
+ "in the audio.cpp catalog (e.g. "
+ "qwen3_tts,higgs_audio_tts). Required in a "
+ "non-interactive run; skips the family tree in "
+ "the TUI")
+ parser.add_argument("--all-packages", action="store_true",
+ help="Host every installable package of each selected "
+ "family (distinct target_directory) instead of "
+ "only the recommended one. Voice-design packages "
+ "are hosted with task 'vdes'")
+ parser.add_argument("--host", type=str, default=None,
+ help="Bind host for the server (default: 127.0.0.1)")
+ parser.add_argument("--port", type=int, default=None,
+ help="Port for the server (default: the port in "
+ "AUDIOCPP_API_URL from app/converter/config.py)")
+ parser.add_argument("--backend", choices=BACKENDS, default=None,
+ help="Inference backend recorded in server.json "
+ "(default: auto-detected from the checkout's "
+ "build/ directory, else cuda)")
+ parser.add_argument("--build-backend", choices=BACKENDS, default=None,
+ help="Build audiocpp_server for this backend when it "
+ "is not built yet, and use it in server.json")
+ parser.add_argument("--whisper-model", type=str, default="base",
+ help="Whisper model size for transcription "
+ "(default: base)")
+ parser.add_argument("--force", action="store_true",
+ help="Overwrite the output file (and prompt_text) "
+ "without prompting; in the TUI, start the "
+ "wizard fresh instead of loading the existing "
+ "server.json")
+ parser.add_argument("--download", action="store_true",
+ help="Run model_manager_v2.py install for each hosted "
+ "model automatically (default: print the commands "
+ "only)")
+ parser.add_argument("--no-sync-port", action="store_true",
+ help="Do not rewrite AUDIOCPP_API_URL in "
+ "app/converter/config.py when --port differs")
+ parser.add_argument("--no-sync-model-ids", action="store_true",
+ help="Do not rewrite AUDIOCPP_MODEL_ID/"
+ "AUDIOCPP_CLONE_MODEL_ID for a single-entry server")
+ return parser
+
+
+def main() -> int:
+ parser = build_parser()
+ args = parser.parse_args()
+
+ if args.input_dir is not None and not args.input_dir.is_dir():
+ parser.error(
+ f"WAV directory not found: {args.input_dir}\n"
+ f" (resolved from the current working directory: "
+ f"{Path.cwd()})\n"
+ " --wavs must be a directory containing the .wav "
+ "reference files to use as voice cloning presets")
+
+ if _interactive():
+ return run_tui(args, parser)
+
+ # Non-interactive (no terminal, or all flags supplied): flag-only path.
+ settings = _collect_from_flags(args, parser)
+ if settings is None:
+ return 1
+ return _execute(settings, args)
+
+