diff options
| author | historia <historiavg@proton.me> | 2026-08-26 02:25:55 -0400 |
|---|---|---|
| committer | historia <historiavg@proton.me> | 2026-08-26 02:25:55 -0400 |
| commit | 8b5c8697740ff415cf7f1d03c9fb5a8c8851d420 (patch) | |
| tree | 28c0323c54c896af5f89fb34b89a62e0fe0df291 /app/backends/audiocpp | |
| parent | acbd9ff2c91182d96c57ffb57bee6e9b3fcbcbd4 (diff) | |
| download | tts-audiobook-generator-8b5c8697740ff415cf7f1d03c9fb5a8c8851d420.tar.gz | |
refactor: audiocpp.py setup flow
Diffstat (limited to 'app/backends/audiocpp')
| -rw-r--r-- | app/backends/audiocpp/__init__.py | 107 | ||||
| -rw-r--r-- | app/backends/audiocpp/__main__.py | 8 | ||||
| -rw-r--r-- | app/backends/audiocpp/build.py | 339 | ||||
| -rw-r--r-- | app/backends/audiocpp/catalog.py | 293 | ||||
| -rw-r--r-- | app/backends/audiocpp/configsync.py | 163 | ||||
| -rw-r--r-- | app/backends/audiocpp/constants.py | 32 | ||||
| -rw-r--r-- | app/backends/audiocpp/models.py | 384 | ||||
| -rw-r--r-- | app/backends/audiocpp/patches/ggml-top-k-cuda-iterator.patch | 10 | ||||
| -rw-r--r-- | app/backends/audiocpp/remote.py | 59 | ||||
| -rw-r--r-- | app/backends/audiocpp/status.py | 90 | ||||
| -rw-r--r-- | app/backends/audiocpp/voices.py | 146 | ||||
| -rw-r--r-- | app/backends/audiocpp/wizard.py | 1081 |
12 files changed, 2712 insertions, 0 deletions
diff --git a/app/backends/audiocpp/__init__.py b/app/backends/audiocpp/__init__.py new file mode 100644 index 0000000..84166af --- /dev/null +++ b/app/backends/audiocpp/__init__.py @@ -0,0 +1,107 @@ +"""audio.cpp backend — setup wizard, server config, model management. + +A package of focused modules behind one import surface: everything public +is re-exported here, so callers (the hub, the CLI, tests) keep using +``backends.audiocpp.<name>`` regardless of which module implements it. + +Modules: + constants shared constants (backends, tasks, checkout location) + catalog the model_specs catalog + server.json building/selections + models install state on disk, missing-model guidance, downloads + voices reference-.wav transcription planning and execution + configsync app/converter/config.py + server.json port/id/backend sync + build checkout lifecycle: ggml patches, binary build, uninstall + remote querying a running server for its models/voices + status detect() for the hub's backend menu + wizard the TUI wizard and the CLI entry points +""" + +from .constants import ( + AUDIOCPP_DIR_NAME, + AUDIOCPP_GIT_URL, + BACKENDS, + DEFAULT_HOST, + FALLBACK_PORT, + PATCH_DIR, + TASK_TTS, + TASK_VDES, +) +from .catalog import ( + _backend_options, + _default_package, + detect_backend, + is_design_package, + load_model_catalog, + package_dir_options, + build_model_entry, + build_server_config, + load_server_config, + server_config_selections, +) +from .models import ( + delete_model_files, + hand_install_guidance, + install_models, + installed_model_entries, + missing_model_entries, + missing_model_install_guidance, + model_install_hints, + unused_installed_entries, +) +from .voices import ( + print_empty_transcript_warning, + transcribe_wav_dir, +) +from .configsync import ( + config_port, + update_config_api_url_port, + update_config_model_ids, + update_server_backend, + update_server_config_port, +) +from .build import ( + apply_ggml_patches, + build_audiocpp, + find_build_script, + find_local_checkout, + find_audiocpp_server_bin, + uninstall, +) +from .remote import fetch_server_models, fetch_server_voices +from .wizard import ( + build_parser, + build_screen, + main, + run_tui, + setup_screen, +) +from .status import detect + +__all__ = [ + # constants + "AUDIOCPP_DIR_NAME", "AUDIOCPP_GIT_URL", "BACKENDS", "DEFAULT_HOST", + "FALLBACK_PORT", "PATCH_DIR", "TASK_TTS", "TASK_VDES", + # catalog + "detect_backend", "load_model_catalog", "is_design_package", + "package_dir_options", "build_model_entry", "build_server_config", + "load_server_config", "server_config_selections", + # models + "missing_model_entries", "installed_model_entries", + "unused_installed_entries", "delete_model_files", "install_models", + "hand_install_guidance", "missing_model_install_guidance", + "model_install_hints", + # voices + "transcribe_wav_dir", "print_empty_transcript_warning", + # configsync + "config_port", "update_config_api_url_port", + "update_config_model_ids", "update_server_config_port", + "update_server_backend", + # build + "find_local_checkout", "find_audiocpp_server_bin", "find_build_script", + "apply_ggml_patches", "build_audiocpp", "uninstall", + # remote + "fetch_server_models", "fetch_server_voices", + # wizard / status + "setup_screen", "build_screen", "run_tui", "build_parser", "detect", + "main", +] diff --git a/app/backends/audiocpp/__main__.py b/app/backends/audiocpp/__main__.py new file mode 100644 index 0000000..457cddc --- /dev/null +++ b/app/backends/audiocpp/__main__.py @@ -0,0 +1,8 @@ +"""Direct CLI execution: ``python -m backends.audiocpp`` (from app/).""" + +import sys + +from backends.audiocpp import main + +if __name__ == "__main__": + sys.exit(main()) diff --git a/app/backends/audiocpp/build.py b/app/backends/audiocpp/build.py new file mode 100644 index 0000000..a4b307a --- /dev/null +++ b/app/backends/audiocpp/build.py @@ -0,0 +1,339 @@ +"""Checkout lifecycle: clone location, ggml patches, binary build, uninstall.""" + +import contextlib +import io +import re +import shlex +import shutil +from datetime import datetime +from pathlib import Path +from typing import List, Optional + +from backends import common, servers +from backends.common import APP_DIR +from .catalog import _BACKEND_TOKEN_RE +from .constants import ( + AUDIOCPP_DIR_NAME, + AUDIOCPP_GIT_URL, + PATCH_DIR, +) + +def uninstall(*, emit=None, cancel=None) -> int: + """Remove the audio.cpp backend entirely: stop its server, delete the checkout. + + The checkout (``app/audio.cpp``) holds the built binary, the downloaded + models, and the server.json, so removing the directory uninstalls the + backend. A running server this tool started is stopped first + (best-effort). + + EMIT is accepted for registry symmetry with the other backends but is + unused here — this uninstall has no subprocess phase, and its prints are + captured by the task view when run in the TUI. CANCEL is a + ``threading.Event`` honored between phases only (after the server has + been stopped, before the checkout is deleted), so a started phase always + completes and the uninstall never tears halfway. Returns the exit code + (130 when cancelled before a remaining phase). + """ + # Only stop when a pid file exists: without one this tool never + # started the server, so the "not started by this tool" notice would + # be uninstall-time noise. + if servers.pid_for("audiocpp") is not None: + servers.stop("audiocpp") + if common.cancel_requested(cancel): + return 130 + checkout = find_local_checkout() + if checkout is None: + print("[INFO] No audio.cpp checkout to remove.") + return 0 + print(f"[INFO] Removing audio.cpp checkout {checkout}...") + shutil.rmtree(checkout, ignore_errors=True) + print("[OK] audio.cpp removed.") + return 0 + + +def find_local_checkout() -> Optional[Path]: + """Return the managed audio.cpp checkout at ``app/audio.cpp``. + + Returns the path only when it contains a ``model_specs`` directory; + the checkout is installed there by the setup wizard and nowhere else. + """ + try: + resolved = (APP_DIR / AUDIOCPP_DIR_NAME).resolve() + except OSError: + return None + if (resolved / "model_specs").is_dir(): + return resolved + return None + + +def find_audiocpp_server_bin(audiocpp_dir: Path) -> Optional[Path]: + """Return the built audiocpp_server binary, or None when not built. + + Scans ``audiocpp_dir/build/*`` for a build directory containing + ``bin/audiocpp_server`` (``.exe`` allowed on Windows). When several + builds exist the first (alphabetical) is returned. + """ + build_root = audiocpp_dir / "build" + if not build_root.is_dir(): + return None + try: + build_dirs = sorted(build_root.iterdir(), + key=lambda p: p.name.lower()) + except OSError: + return None + for build_dir in build_dirs: + if not build_dir.is_dir(): + continue + for name in ("audiocpp_server", "audiocpp_server.exe"): + server = build_dir / "bin" / name + if server.exists(): + return server + return None + + +def built_server_binary(audiocpp_dir: Path, backend: str) -> Optional[Path]: + """Return the built audiocpp_server for BACKEND, or None. + + Like ``find_audiocpp_server_bin`` but limited to build directories whose + name carries the BACKEND token (``-cuda-``, ``-vulkan-``, ``-hip-``, + ``-cpu-``; ``-metal-`` counts as ``cpu``). A checkout with builds for + several backends is asked which one to use without re-offering a build + for a backend that is already built. + """ + build_root = audiocpp_dir / "build" + if not build_root.is_dir(): + return None + try: + build_dirs = sorted(build_root.iterdir(), + key=lambda p: p.name.lower()) + except OSError: + return None + for build_dir in build_dirs: + if not build_dir.is_dir(): + continue + match = _BACKEND_TOKEN_RE.search(build_dir.name.lower()) + if not match: + continue + token = "cpu" if match.group(1) == "metal" else match.group(1) + if token != backend: + continue + for name in ("audiocpp_server", "audiocpp_server.exe"): + server = build_dir / "bin" / name + if server.exists(): + return server + return None + + +def find_build_script(audiocpp_dir: Path) -> Optional[Path]: + """Return the audio.cpp build helper script to run, or None. + + Prefers ``scripts/build_linux.sh``; otherwise the first + ``scripts/build_*.sh`` it finds. (Windows ``.bat`` scripts are not run + automatically — build manually there.) + """ + scripts = audiocpp_dir / "scripts" + if not scripts.is_dir(): + return None + preferred = scripts / "build_linux.sh" + if preferred.exists(): + return preferred + try: + candidates = sorted(scripts.glob("build_*.sh"), + key=lambda p: p.name.lower()) + except OSError: + return None + return candidates[0] if candidates else None + + +GGML_PATCHES = [ + { + "file": "ggml-top-k-cuda-iterator.patch", + "target": "external/ggml/src/ggml-cuda/top-k.cu", + "marker": r"#\s*include\s*<cuda/iterator>", + "label": "top-k.cu: add #include <cuda/iterator> (CCCL 3.x build fix)", + }, +] + + +def apply_ggml_patches(audiocpp_dir: Path, *, emit=None, cancel=None) -> int: + """Apply the shipped ggml build patches to an audio.cpp checkout. + + Idempotent: a patch whose marker already matches its target is skipped + (it is either already applied, or the fork re-vendored a fixed ggml). A + patch that no longer applies because the vendored file changed shape is a + loud, non-interactive failure — the build is aborted so the user + re-evaluates the patch instead of hitting a known nvcc break minutes + later. Returns 0 when every patch is applied or already present, 1 on + drift, 130 when cancelled. + """ + for patch in GGML_PATCHES: + if cancel is not None and cancel.is_set(): + return 130 + target = audiocpp_dir / patch["target"] + if not target.is_file(): + print(f"[INFO] {patch['file']}: target {patch['target']} not " + f"present in this checkout; skipping") + continue + try: + text = target.read_text(encoding="utf-8", errors="ignore") + except OSError as exc: + print(f"[WARNING] {patch['file']}: could not read {target}: " + f"{exc}; skipping") + continue + if re.search(patch["marker"], text): + print(f"[OK] {patch['file']}: fix already present, skipping") + continue + patch_path = PATCH_DIR / patch["file"] + if not patch_path.is_file(): + print(f"[ERROR] {patch['file']}: patch file not found at " + f"{patch_path}; cannot apply") + return 1 + check_argv = ["git", "-C", str(audiocpp_dir), "apply", "--check", + "--whitespace=nowarn", str(patch_path)] + check_rc = common.run_console_subprocess( + check_argv, emit=emit, cancel=cancel) + if check_rc == 130 or (cancel is not None and cancel.is_set()): + return 130 + if check_rc != 0: + print(f"[ERROR] {patch['file']}: no longer applies to " + f"{patch['target']} (git apply --check exit {check_rc}). " + f"The audio.cpp fork's vendored ggml changed shape and " + f"still lacks the fix. Re-evaluate {patch_path}: " + f"regenerate the patch, or drop this entry if the fork " + f"now ships the fix.") + return 1 + apply_argv = ["git", "-C", str(audiocpp_dir), "apply", + "--whitespace=nowarn", str(patch_path)] + rc = common.run_console_subprocess( + apply_argv, emit=emit, cancel=cancel) + if rc == 130 or (cancel is not None and cancel.is_set()): + return 130 + if rc != 0: + print(f"[ERROR] {patch['file']}: git apply failed (exit {rc})") + return rc + print(f"[OK] {patch['file']}: applied ({patch['label']})") + return 0 + + +def build_audiocpp(audiocpp_dir: Path, backend: str, *, + emit=None, cancel=None) -> int: + """Build audiocpp_server for BACKEND, streaming output. + + With EMIT None the build script runs on the console (inherits the + terminal); with EMIT given (the in-TUI task view) its output streams line + by line to EMIT so the view can show progress, and CANCEL aborts it. + + On the EMIT (TUI) path the build output is also tee'd to + ``app/logs/audiocpp_build_<timestamp>.log`` so it survives the curses + session; when the build fails (and was not cancelled) a post-TUI notice + with the copy-pastable command and the log path is queued for the console + (see ``backends.common.record_post_tui_notice``). + + Returns the build script's exit code (non-zero when the script is + missing). + """ + script = find_build_script(audiocpp_dir) + if script is None: + message = (f"[ERROR] No build script found in {audiocpp_dir}/scripts; " + "build audiocpp_server manually (see the audio.cpp README)") + print(message) + if emit is not None: + common.record_post_tui_notice(message) + return 1 + argv = ["sh", str(script), "--backend", backend, "--target", + "audiocpp_server", "--deployment-build"] + command = f"cd {audiocpp_dir} && {shlex.join(argv)}" + if emit is None: + print(f"[INFO] Building audiocpp_server for {backend} ({command})...") + patch_rc = apply_ggml_patches(audiocpp_dir, cancel=cancel) + if patch_rc == 130 or (cancel is not None and cancel.is_set()): + return 130 + if patch_rc != 0: + print("[ERROR] ggml build patches could not be applied; " + "aborting audiocpp_server build. See the messages above " + "and re-evaluate app/backends/patches/.") + return patch_rc + return common.run_console_subprocess(argv, cwd=audiocpp_dir) + return _build_audiocpp_tui(emit, cancel, argv, command, audiocpp_dir) + + +def _build_audiocpp_tui(emit, cancel, argv: List[str], command: str, + audiocpp_dir: Path) -> int: + """Run the build on the TUI path: tee output to a log file. + + The ggml patch step runs first, inside the same log: every emitted + line (patch status, build output) is also written (and flushed) to + ``app/logs/audiocpp_build_<timestamp>.log``. On failure a summary (the + copy-pastable COMMAND and the log path) is emitted into the TUI, + written to the log, and queued as a post-TUI console notice. A + cancelled build (CANCEL set) is not reported as a failure, but its + partial output stays in the log file. + """ + log_path = common.LOG_DIR / ( + f"audiocpp_build_{datetime.now():%Y%m%d_%H%M%S}.log") + log_path.parent.mkdir(parents=True, exist_ok=True) + log_handle = log_path.open("w", encoding="utf-8") + + def tee(line: str) -> None: + log_handle.write(line + "\n") + log_handle.flush() + emit(line) + + class _TeeWriter(io.TextIOBase): + """Route print() output from the patch step into the log too.""" + + def write(self, s: str) -> int: + for line in s.splitlines(): + if line: + tee(line) + return len(s) + + try: + with contextlib.redirect_stdout(_TeeWriter()): + patch_rc = apply_ggml_patches(audiocpp_dir, emit=tee, + cancel=cancel) + if patch_rc == 130 or (cancel is not None and cancel.is_set()): + return 130 + if patch_rc != 0: + notice = ("[ERROR] ggml build patches could not be applied; " + "aborting audiocpp_server build. See the messages " + "above and re-evaluate app/backends/patches/.") + tee(notice) + common.record_post_tui_notice(notice) + return patch_rc + tee(f"[INFO] Building audiocpp_server ({command})...") + rc = common.run_console_subprocess( + argv, cwd=audiocpp_dir, emit=tee, cancel=cancel) + if rc != 0 and (cancel is None or not cancel.is_set()): + notice = (f"[ERROR] audio.cpp build failed (exit code {rc}).\n" + f" Build log: {log_path}\n" + f" Troubleshoot by re-running this command:\n" + f" {command}") + for line in notice.splitlines(): + tee(line) + common.record_post_tui_notice(notice) + finally: + log_handle.close() + return rc + + +def _print_launch_hint(audiocpp_dir: Path, output_path: Path) -> None: + """Print remediation when audiocpp_server is missing (troubleshooting). + + The hub starts and stops the server itself, so a working install gets + no manual launch instructions. When no binary was built, though, the + user needs to know how to build and run it by hand. The commands are + prefixed with ``cd <checkout> &&`` because the server discovers + model_specs/<family>.json relative to its working directory. + """ + if find_audiocpp_server_bin(audiocpp_dir) is not None: + return + print("\n[INFO] audiocpp_server binary not found. Build it first, e.g.:") + script = find_build_script(audiocpp_dir) + if script is not None: + print(f" sh {script} --backend <cuda|vulkan|hip|cpu> " + "--target audiocpp_server --deployment-build") + print(f" then run: cd {audiocpp_dir} && ./build/<platform>-<backend>" + f"-release/bin/audiocpp_server --config {output_path}") + + diff --git a/app/backends/audiocpp/catalog.py b/app/backends/audiocpp/catalog.py new file mode 100644 index 0000000..c672232 --- /dev/null +++ b/app/backends/audiocpp/catalog.py @@ -0,0 +1,293 @@ +"""The model_specs catalog, server.json building and selection views.""" + +import json +import re +from pathlib import Path +from typing import Dict, List, Optional, Set, Tuple + +from .. import common +from .constants import ( + BACKENDS, + DEFAULT_HOST, + FALLBACK_PORT, + TASK_TTS, +) + +DESIGN_PACKAGE_RE = re.compile(r"voice[\s_\-]?design", re.IGNORECASE) + + +_BACKEND_DESCRIPTIONS = ( + ("cuda", "NVIDIA GPUs (fastest)"), + ("vulkan", "cross-vendor GPU"), + ("hip", "AMD GPUs"), + ("cpu", "no GPU required"), +) + + +def _backend_options(detected: Optional[str] = None + ) -> Tuple[List[Tuple[str, str]], int]: + """Build the aligned backend menu options and the default index. + + The backend names are padded to a common width so the ``-`` dashes + before the descriptions line up. When DETECTED matches one of the + options, that option gets ``[auto-detected]`` appended and is the + default (cursor/start) selection; otherwise the first option is the + default as before. Returns (options, default_index). + """ + width = max(len(name) for name, _ in _BACKEND_DESCRIPTIONS) + options: List[Tuple[str, str]] = [] + default_index = 0 + for index, (name, desc) in enumerate(_BACKEND_DESCRIPTIONS): + label = f"{name.ljust(width)} - {desc}" + if detected == name: + label += " [auto-detected]" + default_index = index + options.append((label, name)) + return options, default_index + + +_BACKEND_TOKEN_RE = re.compile(r"-(cuda|vulkan|hip|cpu|metal)(?:-|$)") + + +def detect_backend(audiocpp_dir: Path) -> Optional[str]: + """Best-effort detection of the backend audiocpp_server was built for. + + Scans ``audiocpp_dir/build/*`` for build directories that contain a + built ``bin/audiocpp_server`` (``.exe`` allowed on Windows) and reads + the backend token out of the directory name (``-cuda-``, ``-vulkan-``, + ``-hip-`` or ``-cpu-``; ``-metal-`` is mapped to ``cpu``). Returns the + backend only when exactly one distinct backend was built, so a checkout + with builds for several backends does not silently pick one. Returns + None when there is no ``build/`` directory, no built server, or more + than one distinct backend. + """ + build_root = audiocpp_dir / "build" + if not build_root.is_dir(): + return None + backends: Set[str] = set() + try: + build_dirs = sorted(build_root.iterdir(), + key=lambda p: p.name.lower()) + except OSError: + return None + for build_dir in build_dirs: + if not build_dir.is_dir(): + continue + server = build_dir / "bin" / "audiocpp_server" + if not server.exists(): + server_exe = build_dir / "bin" / "audiocpp_server.exe" + if not server_exe.exists(): + continue + match = _BACKEND_TOKEN_RE.search(build_dir.name.lower()) + if not match: + continue + token = match.group(1) + backends.add("cpu" if token == "metal" else token) + if len(backends) == 1: + return next(iter(backends)) + return None + + +def _default_package(packages: List[dict]) -> Optional[dict]: + """Pick the default package from a list of packages. + + Prefers the package flagged ``default: true``, then the first GGUF + package, then the first package overall. Returns None for an empty list. + """ + if not packages: + return None + for package in packages: + if package.get("default"): + return package + for package in packages: + if package.get("format") == "gguf": + return package + return packages[0] + + +def load_model_catalog(audiocpp_dir: Path) -> List[dict]: + """Read model_specs/*.json and return the TTS-capable families. + + Each returned entry has: family, display_name, description, languages, + clone_capable, packages (the full list from the spec), install_id + (recommended package id), and default_path (``models/<target_directory>``). + All families are treated equally and listed in alphabetical order by + display name. + """ + specs_dir = audiocpp_dir / "model_specs" + if not specs_dir.is_dir(): + raise NotADirectoryError( + f"{audiocpp_dir} has no model_specs/ directory; re-run setup " + "to refresh the audio.cpp checkout") + entries: List[dict] = [] + for spec_path in sorted(specs_dir.glob("*.json")): + try: + spec = json.loads(spec_path.read_text(encoding="utf-8")) + except (OSError, ValueError): + continue + tasks = spec.get("tasks") or [] + if "tts" not in tasks and spec.get("category") != "tts": + continue + family = spec.get("family") or spec_path.stem + packages = spec.get("packages") or [] + package = _default_package(packages) + if package is None: + # No installable package: skip (cannot be hosted from a path). + continue + target_directory = package.get("target_directory") or family + languages = spec.get("languages") or [] + display_name = spec.get("display_name") or family + description = spec.get("description") or "" + entries.append({ + "family": family, + "display_name": display_name, + "description": description, + "languages": languages, + "tasks": list(tasks), + "clone_capable": "clone" in tasks, + "packages": packages, + "install_id": package.get("id") or family, + "default_path": f"models/{target_directory}", + }) + + # All families are treated equally: alphabetical by display name. + entries.sort(key=lambda entry: entry["display_name"].lower()) + return entries + + +def is_design_package(package: dict) -> bool: + """Return True when a package's name marks it a voice-design model. + + audio.cpp voice-design packages (whose id, display name, or target + directory mentions "voice design") are the only packages that must be + hosted with task "vdes"; their role is not in the schema, only in those + strings, so it is detected from them. + """ + text = " ".join(str(package.get(key, "")) + for key in ("id", "display_name", "target_directory")) + return bool(DESIGN_PACKAGE_RE.search(text)) + + +def package_dir_options(entry: dict) -> List[dict]: + """Return one option per distinct target_directory of a family's packages. + + Each option is a dict with: target_directory, install_id (the recommended + package id inside that directory), design (voice-design package flag), and + recommended (whether it holds the family's default package). Precisions + that share a directory (q8_0/bf16/...) collapse to a single option. + """ + packages = entry.get("packages") or [] + default_pkg = _default_package(packages) + default_dir = (default_pkg or {}).get("target_directory") or entry["family"] + by_dir: Dict[str, List[dict]] = {} + order: List[str] = [] + for package in packages: + directory = package.get("target_directory") or entry["family"] + if directory not in by_dir: + by_dir[directory] = [] + order.append(directory) + by_dir[directory].append(package) + options: List[dict] = [] + for directory in order: + package = _default_package(by_dir[directory]) + options.append({ + "target_directory": directory, + "install_id": (package or {}).get("id") or directory, + "design": is_design_package(package or {}), + "recommended": directory == default_dir, + }) + # Put the recommended package first for a friendlier checklist. + options.sort(key=lambda opt: not opt["recommended"]) + return options + + +def build_model_entry(family: str, model_id: str, model_path: str, + task: str = TASK_TTS) -> dict: + """Assemble one server.json model entry. + + ``task`` defaults to "tts"; voice design packages are hosted with + "vdes" so the server runs its design session for speech requests + (audiobook.py then requires --instructions with that entry). + """ + return { + "id": model_id, + "family": family, + "path": model_path, + "task": task, + "mode": "offline", + } + + +def build_server_config(host: str, port: int, backend: str, lazy_load: bool, + model_entries: List[dict], + voice_dir: Optional[str] = None) -> dict: + """Assemble the server.json document. + + ``voice_dir`` is a server-level cloning voice library; when set, every + hosted clone-capable family can use its voices with ``--voice``. + """ + config_doc = { + "host": host, + "port": port, + "backend": backend, + "lazy_load": lazy_load, + "models": model_entries, + } + if voice_dir: + config_doc["voice_dir"] = voice_dir + return config_doc + + +def load_server_config(server_json: Path) -> Optional[dict]: + """Read server.json into a dict, or None when it cannot be used. + + Returns None for a missing file, unreadable content, or a non-dict + document. Used by the wizard's modify flow to pre-fill its screens + from an existing config instead of prompting to overwrite it. + """ + if not server_json.exists(): + return None + try: + data = json.loads(server_json.read_text(encoding="utf-8")) + except (OSError, ValueError): + return None + if not isinstance(data, dict): + return None + return data + + +def server_config_selections(server_config: dict, + catalog: List[dict] + ) -> Tuple[Dict[str, List[str]], + Dict[Tuple[str, str], str]]: + """Map an existing server.json's models back to catalog selections. + + Returns ``(selected_dirs, tasks)``: ``selected_dirs`` maps a catalog + family to the target directories it hosts (``models/<target>`` paths + with the ``models/`` prefix stripped, in server.json order), and + ``tasks`` maps ``(family, target_directory)`` to the entry's task + (``"tts"`` or ``"vdes"``) so the wizard can preserve how design + packages were hosted. Entries whose family is not in the CATALOG are + ignored — the wizard cannot offer them again. + """ + families = {entry["family"] for entry in catalog} + selected_dirs: Dict[str, List[str]] = {} + tasks: Dict[Tuple[str, str], str] = {} + for entry in server_config.get("models") or []: + if not isinstance(entry, dict): + continue + family = entry.get("family") + if not isinstance(family, str) or family not in families: + continue + path = entry.get("path") + if not isinstance(path, str): + continue + target = path[len("models/"):] if path.startswith("models/") else path + if family not in selected_dirs: + selected_dirs[family] = [] + if target not in selected_dirs[family]: + selected_dirs[family].append(target) + tasks[(family, target)] = str(entry.get("task") or TASK_TTS) + return selected_dirs, tasks + + diff --git a/app/backends/audiocpp/configsync.py b/app/backends/audiocpp/configsync.py new file mode 100644 index 0000000..0a161c0 --- /dev/null +++ b/app/backends/audiocpp/configsync.py @@ -0,0 +1,163 @@ +"""Keep app/converter/config.py and server.json in sync with setup choices.""" + +import json +import re +import urllib.parse +from pathlib import Path +from typing import Optional + +from backends import common +from backends.common import CONFIG_PATH, url_with_port +from converter import config +from . import build +from .constants import FALLBACK_PORT + +def config_port() -> int: + """Return the port of AUDIOCPP_API_URL in app/converter/config.py.""" + try: + return urllib.parse.urlsplit(config.AUDIOCPP_API_URL).port or FALLBACK_PORT + except ValueError: + return FALLBACK_PORT + + +def update_config_api_url_port(port: int, config_path: Optional[Path] = None) -> bool: + """Rewrite the port inside AUDIOCPP_API_URL in app/converter/config.py. + + Reads the configured URL from the file (not from the imported module, + which a long hub session can leave behind), swaps its port for PORT, + and writes it back through ``common.update_config_value`` so the + imported module mirrors the change immediately. Returns True when the + file now holds the new URL. + """ + path = Path(config_path) if config_path is not None else CONFIG_PATH + try: + text = path.read_text(encoding="utf-8") + except OSError: + return False + match = re.search(r'(?m)^\s*AUDIOCPP_API_URL\s*=\s*"([^"]*)"', text) + if not match: + return False + return common.update_config_value("AUDIOCPP_API_URL", + url_with_port(match.group(1), port), + config_path=path) + + +def update_server_config_port(port: int) -> bool: + """Rewrite the 'port' in the audio.cpp checkout's server.json. + + Loads ``<checkout>/server.json``, sets its ``port`` to PORT, and + rewrites it with the same ``json.dump`` formatting the wizard uses. + Returns True when the file now carries PORT (a no-op when it already + does), and False when there is no checkout/server.json or the file + cannot be read or written. + """ + checkout = build.find_local_checkout() + if checkout is None: + return False + server_json = checkout / "server.json" + if not server_json.exists(): + return False + try: + data = json.loads(server_json.read_text(encoding="utf-8")) + except (OSError, ValueError): + return False + if not isinstance(data, dict): + return False + if data.get("port") == port: + return True + data["port"] = port + try: + with server_json.open("w", encoding="utf-8") as handle: + json.dump(data, handle, indent=2, ensure_ascii=False) + handle.write("\n") + except OSError: + return False + return True + + +def update_config_model_ids(model_id: str, + clone_model_id: Optional[str] = None, + config_path: Optional[Path] = None) -> bool: + """Rewrite AUDIOCPP_MODEL_ID (and AUDIOCPP_CLONE_MODEL_ID when given). + + Goes through ``common.update_config_value`` so the imported config + module mirrors the change immediately. Returns True when every named + key now holds its value in the file. + """ + path = Path(config_path) if config_path is not None else CONFIG_PATH + ok = common.update_config_value("AUDIOCPP_MODEL_ID", model_id, + config_path=path) + if clone_model_id is not None: + ok = common.update_config_value("AUDIOCPP_CLONE_MODEL_ID", + clone_model_id, + config_path=path) and ok + return ok + + +def _apply_port_sync(port: int, accepted: bool) -> None: + """Write the port into app/converter/config.py, or report when declined.""" + if accepted: + if not update_config_api_url_port(port): + print(f"[WARNING] Could not update {CONFIG_PATH}; edit " + "AUDIOCPP_API_URL by hand so audiobook.py uses the " + "new port") + else: + print("[WARNING] Left AUDIOCPP_API_URL unchanged; audiobook.py " + f"will still use port {config_port()}") + + +def _offer_config_model_id_sync(model_id: str, accepted: Optional[bool]) -> None: + """Point app/converter/config.py at a single hosted model entry. + + The converter requests the model id configured in AUDIOCPP_MODEL_ID, + and single-model servers use the same id for the clone entry, so both + ids are rewritten together. ACCEPTED is True/False (apply/skip the + rewrite) or None when no single-entry sync applies (nothing to do). + """ + if config.AUDIOCPP_MODEL_ID == model_id \ + and config.AUDIOCPP_CLONE_MODEL_ID == model_id: + return + if accepted is None: + return + if accepted: + if not update_config_model_ids(model_id, model_id): + print(f"[WARNING] Could not update {CONFIG_PATH}; edit " + "AUDIOCPP_MODEL_ID and AUDIOCPP_CLONE_MODEL_ID by hand so " + "audiobook.py uses this model") + else: + print("[WARNING] Left the model ids unchanged; audiobook.py will " + f"still request model '{config.AUDIOCPP_MODEL_ID}'") + + +def update_server_backend(backend: str) -> bool: + """Rewrite the 'backend' in the checkout's server.json, or True when none. + + Sets ``backend`` to BACKEND in ``<checkout>/server.json`` (same + ``json.dump`` formatting as the wizard). Returns True when the file now + carries BACKEND, when there is no server.json (nothing to sync), or when + it already does; False when the file exists but cannot be read/written. + """ + checkout = build.find_local_checkout() + if checkout is None: + return True + server_json = checkout / "server.json" + if not server_json.exists(): + return True + try: + data = json.loads(server_json.read_text(encoding="utf-8")) + except (OSError, ValueError): + return False + if not isinstance(data, dict): + return False + if data.get("backend") == backend: + return True + data["backend"] = backend + try: + with server_json.open("w", encoding="utf-8") as handle: + json.dump(data, handle, indent=2, ensure_ascii=False) + handle.write("\n") + except OSError: + return False + return True + + diff --git a/app/backends/audiocpp/constants.py b/app/backends/audiocpp/constants.py new file mode 100644 index 0000000..aaa1eed --- /dev/null +++ b/app/backends/audiocpp/constants.py @@ -0,0 +1,32 @@ +"""Constants shared across the audio.cpp backend modules.""" + +import re +from pathlib import Path + +DEFAULT_HOST = "127.0.0.1" + + +FALLBACK_PORT = 8080 + + +BACKENDS = ("cuda", "vulkan", "hip", "cpu") + + +TASK_TTS = "tts" + + +TASK_VDES = "vdes" + + +AUDIOCPP_DIR_NAME = "audio.cpp" + + +AUDIOCPP_GIT_URL = "https://github.com/0xShug0/audio.cpp" + + + +# ggml build patches shipped in this repo and applied to the (gitignored) +# audio.cpp checkout before building, so a fresh clone survives known ggml +# build bugs the audio.cpp fork has not re-vendored yet. See +# apply_ggml_patches() in backends.audiocpp.build. +PATCH_DIR = Path(__file__).resolve().parent / "patches" diff --git a/app/backends/audiocpp/models.py b/app/backends/audiocpp/models.py new file mode 100644 index 0000000..4e6b8bb --- /dev/null +++ b/app/backends/audiocpp/models.py @@ -0,0 +1,384 @@ +"""Model install state: what is on disk, what is missing, how to fetch it.""" + +import json +import os +import shutil +import sys +import tempfile +from pathlib import Path +from typing import Callable, Dict, List, Optional, Set, Tuple + +from backends import common +from . import catalog as _catalog + +def _install_models(audiocpp_dir: Path, + install_guidance: List[Tuple[str, str]], + download: bool, emit=None, cancel=None) -> int: + """Print and optionally run the model install commands. + + One ``python <manager> install <id>`` command per hosted model (de-duped + by install id). When DOWNLOAD is True each command is run in the audio.cpp + checkout via ``subprocess`` so the models are downloaded automatically; + a failing install is reported as a warning and does not abort the + remaining downloads. When DOWNLOAD is False (or the model manager is + missing) the commands are only printed, copy-pasteable as before. + + With EMIT given (the in-TUI task view) each download streams its output + to EMIT and — when the checkout's ``model_manager_v2.py`` supports it — + runs with ``--progress --cancel-file`` so the view can show a real byte + progress bar and cancel gracefully. CANCEL aborts a running download. + + Returns 0 when every command succeeded (or nothing needed running), + 130 when cancelled, 1 when any download failed. + """ + manager = audiocpp_dir / "tools" / "model_manager_v2.py" + seen: Set[str] = set() + install_ids: List[str] = [] + for _, install_id in install_guidance: + if install_id not in seen: + seen.add(install_id) + install_ids.append(install_id) + + supports_progress = emit is not None and _manager_supports_progress(manager) + + if download and not manager.is_file(): + print(f"[WARNING] {manager} not found; printing the install commands " + "instead of running them") + download = False + + failed = False + for install_id in install_ids: + command = f"python {manager} install {install_id}" + if not download: + print(command) + continue + print(f"[INFO] Downloading {install_id}...") + argv = [sys.executable, str(manager), "install", install_id] + cancel_file: Optional[Path] = None + on_cancel = None + if supports_progress: + fd, cancel_path = tempfile.mkstemp( + prefix="audiocpp_cancel_", suffix=".cancel") + os.close(fd) + cancel_file = Path(cancel_path) + cancel_file.unlink() # absent = not cancelled + argv += ["--progress", "--cancel-file", str(cancel_file)] + on_cancel = cancel_file.touch + try: + rc = common.run_console_subprocess( + argv, cwd=str(audiocpp_dir), emit=emit, cancel=cancel, + on_cancel=on_cancel) + except OSError as exc: + print(f"[WARNING] Could not run {command}: {exc}") + rc = 1 + finally: + if cancel_file is not None: + try: + cancel_file.unlink() + except OSError: + pass + if rc == 130 or (cancel is not None and cancel.is_set()): + return 130 + if rc != 0: + failed = True + print(f"[WARNING] install {install_id} exited with code " + f"{rc}; the model may need to be downloaded " + "by hand") + return 1 if failed else 0 + + +def _manager_supports_progress(manager: Path) -> bool: + """True when MANAGER (model_manager_v2.py) supports --progress output. + + The ``--progress``/``--cancel-file`` flags are relatively recent; an + older audio.cpp checkout may not have them, so probe the script source + once instead of failing the download with an unknown flag. + """ + try: + text = manager.read_text(encoding="utf-8", errors="ignore") + except OSError: + return False + return "AUDIOCPP_PROGRESS" in text and "--cancel-file" in text + + +def _decide_download(audiocpp_dir: Path, + model_entries: List[dict], + confirm: Callable[[str, bool], bool]) -> bool: + """Ask whether to download the selected models now. + + CONFIRM asks the yes/no question (ask_bool for the line prompts, a TUI + confirm for the wizard). When the audio.cpp model manager is missing the + prompt is skipped and False is returned, so the install commands are only + printed rather than offered to run. The prompt is also skipped (False) + when every selected model is already on disk (see ``_all_models_present``), + so an already-configured checkout is not asked to re-download models it + already has. + """ + manager = audiocpp_dir / "tools" / "model_manager_v2.py" + if not manager.is_file(): + return False + if _all_models_present(audiocpp_dir, model_entries): + return False + return confirm( + "Automatically download the selected models with model_manager_v2.py " + "now?", True) + + +def _build_tree_families(catalog: List[dict]) -> List[dict]: + """Shape the catalog into the checkbox_tree widget's family list.""" + families: List[dict] = [] + for entry in catalog: + capabilities = ["tts"] + if "clone" in entry["tasks"]: + capabilities.append("cloning") + if "design" in entry["tasks"]: + capabilities.append("design") + name = entry["display_name"] + options = [] + for opt in _catalog.package_dir_options(entry): + options.append({ + "key": opt["target_directory"], + "label": opt["install_id"], + "recommended": opt["recommended"], + }) + families.append({ + "label": name, + "detail": ", ".join(capabilities), + "options": options, + }) + return families + + +def _model_path_present(path: Path) -> bool: + """True when a server.json model path holds actual model files. + + A present path is either a file (a single-model package) or a non-empty + directory (the usual GGUF package target directory; an empty one means a + download that never ran or was cleaned up halfway). + """ + try: + if path.is_file(): + return True + if path.is_dir(): + return any(path.iterdir()) + except OSError: + return False + return False + + +def _all_models_present(audiocpp_dir: Path, model_entries: List[dict]) -> bool: + """True when every selected model entry's path already holds files on disk. + + Paths resolve against the checkout (where model_manager_v2.py installs + them), honoring absolute paths. Used by the wizard to skip the + "Automatically download the selected models" prompt when nothing is + actually missing. An empty selection is treated as not-present. + """ + if not model_entries: + return False + for entry in model_entries: + rel = entry.get("path") + if not isinstance(rel, str) or not rel: + return False + path = Path(rel) if Path(rel).is_absolute() else audiocpp_dir / rel + if not _model_path_present(path): + return False + return True + + +def missing_model_entries(server_json: Path) -> List[dict]: + """Return the server.json model entries whose files are not on disk. + + Paths resolve exactly like audiocpp_server resolves them (relative paths + against the server.json's directory). Each returned entry carries the + entry ``id`` and ``rel`` (the configured path string); used by ``detect`` + to warn that a conversion would fail until the models are installed. + """ + try: + data = json.loads(server_json.read_text(encoding="utf-8")) + except (OSError, ValueError): + return [] + if not isinstance(data, dict): + return [] + base = server_json.parent + missing: List[dict] = [] + for entry in data.get("models") or []: + if not isinstance(entry, dict): + continue + rel = entry.get("path") + if not isinstance(rel, str) or not rel: + continue + path = Path(rel) if Path(rel).is_absolute() else base / rel + if _model_path_present(path): + continue + missing.append({"id": str(entry.get("id") or rel), "rel": rel}) + return missing + + +def _install_id_by_path(audiocpp_dir: Path) -> Dict[str, str]: + """Map ``models/<target_directory>`` -> catalog install id. + + The catalog package that installs a model is derived from the + ``default_path`` of each TTS family; an entry whose path matches no + catalog package has no install id. + """ + by_path: Dict[str, str] = {} + try: + for entry in _catalog.load_model_catalog(audiocpp_dir): + by_path[entry["default_path"]] = entry["install_id"] + except (NotADirectoryError, OSError): + pass + return by_path + + +def installed_model_entries(server_json: Path) -> List[dict]: + """Return the server.json model entries whose files ARE on disk. + + The complement of ``missing_model_entries``: each returned entry carries + the entry ``id`` and ``rel`` (the configured path string), resolved + exactly like ``missing_model_entries`` (relative against the server.json's + directory). Used by the wizard's "Delete unused models?" step to find + already-downloaded models that were unselected. + """ + try: + data = json.loads(server_json.read_text(encoding="utf-8")) + except (OSError, ValueError): + return [] + if not isinstance(data, dict): + return [] + base = server_json.parent + installed: List[dict] = [] + for entry in data.get("models") or []: + if not isinstance(entry, dict): + continue + rel = entry.get("path") + if not isinstance(rel, str) or not rel: + continue + path = Path(rel) if Path(rel).is_absolute() else base / rel + if _model_path_present(path): + installed.append({"id": str(entry.get("id") or rel), "rel": rel}) + return installed + + +def missing_model_install_guidance(audiocpp_dir: Path, + missing: List[dict]) -> List[Tuple[str, str]]: + """Map MISSING model entries to (display name, install id) pairs. + + The install id is derived from each entry's configured path via the + catalog (see ``_install_id_by_path``); entries whose path matches no + catalog package are skipped (there is no ``model_manager_v2.py install`` + command for them). Feeds ``_install_models`` for the "Download Missing + Models" action. + """ + by_path = _install_id_by_path(audiocpp_dir) + guidance: List[Tuple[str, str]] = [] + for item in missing: + install_id = by_path.get(item["rel"]) + if install_id: + guidance.append((item["id"], install_id)) + return guidance + + +def model_install_hints(audiocpp_dir: Path, + missing: List[dict]) -> List[str]: + """Remediation lines for MISSING model entries (see missing_model_entries). + + Maps each entry's configured path back to the catalog package that + installs it (``models/<target_directory>`` -> install id) so the line + carries the exact ``model_manager_v2.py install`` command; entries whose + directory matches no catalog package just name the path. + """ + by_path = _install_id_by_path(audiocpp_dir) + hints: List[str] = [] + for item in missing: + install_id = by_path.get(item["rel"]) + hint = f"model not downloaded: {item['id']} ({item['rel']})" + if install_id: + hint += (f" — install with: python tools/model_manager_v2.py " + f"install {install_id}") + hints.append(hint) + return hints + + +def install_models(audiocpp_dir: Path, + guidance: List[Tuple[str, str]], + emit=None, cancel=None) -> int: + """Download the (display name, install id) models via the helper script. + + Runs ``model_manager_v2.py install`` for each de-duped install id in the + checkout, streaming to the console (or to EMIT, the in-TUI task view); a + failing install is reported as a warning and does not abort the rest. + Returns 0 when every download succeeded, 130 when cancelled, 1 when any + failed. Used by the hub's "Download Missing Models" action (see + ``missing_model_install_guidance`` for the mapping). + """ + return _install_models(audiocpp_dir, guidance, download=True, + emit=emit, cancel=cancel) + + +def hand_install_guidance(audiocpp_dir: Path, + missing: List[dict]) -> str: + """Explain how to install MISSING model entries by hand. + + Returns a multi-line message listing each missing model's id and the + path its files must be placed in (``rel``, resolved against the + checkout). Used when the missing models cannot be mapped to + a ``model_manager_v2.py install`` command, so the user still knows what + to download and where to put it. + """ + lines = [ + "None of the missing models map to a model_manager_v2.py install " + "command.", + "Download them by hand and place the files at these paths:", + ] + for item in missing: + lines.append(f" {item['id']} -> {item['rel']}") + lines.append(f"(paths are relative to {audiocpp_dir})") + return "\n".join(lines) + + +def unused_installed_entries(server_json: Path, + new_paths: Set[str]) -> List[dict]: + """Return installed server.json entries whose path is not in NEW_PATHS. + + The already-downloaded models (see ``installed_model_entries``) that the + new selection does not host any more — the candidates for the wizard's + "Delete unused models?" prompt. Entries whose files are not on disk are + never listed (there is nothing to delete). + """ + return [entry for entry in installed_model_entries(server_json) + if entry["rel"] not in new_paths] + + +def delete_model_files(server_json: Path, entries: List[dict]) -> int: + """Remove the on-disk model files for ENTRIES ({id, rel}) from disk. + + Each entry's ``rel`` is resolved exactly like the server resolves it + (relative against ``server_json``'s directory; absolute paths honored), + then removed as a directory tree or a single file. Missing entries are + ignored. Returns the number of paths removed. Used by the wizard's + "Delete unused models?" step — the regenerated server.json already only + lists the kept models, so no entry cleanup is needed here. + """ + base = server_json.parent + removed = 0 + for item in entries: + rel = item.get("rel") + if not isinstance(rel, str) or not rel: + continue + path = Path(rel) if Path(rel).is_absolute() else base / rel + try: + if not path.exists(): + continue + if path.is_dir(): + shutil.rmtree(path, ignore_errors=True) + else: + path.unlink() + except OSError as exc: + print(f"[WARNING] Could not remove {path}: {exc}") + continue + print(f"[OK] Removed unused model {path}") + removed += 1 + return removed + + diff --git a/app/backends/audiocpp/patches/ggml-top-k-cuda-iterator.patch b/app/backends/audiocpp/patches/ggml-top-k-cuda-iterator.patch new file mode 100644 index 0000000..0eb89a5 --- /dev/null +++ b/app/backends/audiocpp/patches/ggml-top-k-cuda-iterator.patch @@ -0,0 +1,10 @@ +--- a/external/ggml/src/ggml-cuda/top-k.cu ++++ b/external/ggml/src/ggml-cuda/top-k.cu +@@ -4,6 +4,7 @@ + #ifdef GGML_CUDA_USE_CUB + # include <cub/cub.cuh> + # if (CCCL_MAJOR_VERSION >= 3 && CCCL_MINOR_VERSION >= 2) + # define CUB_TOP_K_AVAILABLE ++# include <cuda/iterator> + using namespace cub; + # endif // CCCL_MAJOR_VERSION >= 3 && CCCL_MINOR_VERSION >= 2 diff --git a/app/backends/audiocpp/remote.py b/app/backends/audiocpp/remote.py new file mode 100644 index 0000000..42b3872 --- /dev/null +++ b/app/backends/audiocpp/remote.py @@ -0,0 +1,59 @@ +"""Query a running audiocpp_server for its models and voices.""" + +import json +import urllib.request +from typing import Dict, List, Optional + +from .constants import FALLBACK_PORT + +def fetch_server_models(api_url: str) -> Optional[List[Dict[str, str]]]: + """List a running audiocpp_server's model entries via GET /v1/models. + + Returns ``[{id, family, task}, ...]`` — the same shape the converter's + client resolves at startup — or None when URL does not answer with a + valid document (wrong server, still starting, older audio.cpp). Used by + the hub to drive the convert menus against a remote server that has no + local server.json describing it. + """ + try: + with urllib.request.urlopen( + f"{api_url.rstrip('/')}/v1/models", timeout=10) as response: + payload = json.loads(response.read().decode("utf-8")) + except (OSError, ValueError): + # URLError/HTTPError/socket errors are OSErrors; a non-JSON body is + # a ValueError. Anything else means "not an audiocpp_server". + return None + entries = payload.get("data") if isinstance(payload, dict) else None + models: List[Dict[str, str]] = [] + for entry in entries or []: + if isinstance(entry, dict) and entry.get("id"): + models.append({ + "id": str(entry["id"]), + "family": str(entry.get("family") or ""), + "task": str(entry.get("task") or ""), + }) + return models + + +def fetch_server_voices(api_url: str, model_id: str) -> Optional[List[str]]: + """List a running audiocpp_server's voices for MODEL_ID. + + Queries ``GET /v1/audio/voices?model=<id>`` — the endpoint the converter + validates ``--voice`` against — and returns its voice-name list, or None + when the server cannot be queried. Lets the hub offer a remote server's + voices without reading its configuration locally. + """ + query = urllib.parse.urlencode({"model": model_id}) + try: + with urllib.request.urlopen( + f"{api_url.rstrip('/')}/v1/audio/voices?{query}", + timeout=10) as response: + payload = json.loads(response.read().decode("utf-8")) + except (OSError, ValueError): + return None + voices = payload.get("voices") if isinstance(payload, dict) else None + if not isinstance(voices, list): + return None + return [str(voice) for voice in voices] + + diff --git a/app/backends/audiocpp/status.py b/app/backends/audiocpp/status.py new file mode 100644 index 0000000..9ccf8cc --- /dev/null +++ b/app/backends/audiocpp/status.py @@ -0,0 +1,90 @@ +"""detect() — the BackendStatus report for the hub's backend menu.""" + +from typing import List, Tuple + +from backends import (BackendStatus, ServerSpec, format_launch_hint, + probe, servers) +from converter import config +from . import build as _build +from . import models as _models + +def detect() -> BackendStatus: + """Detect how far audio.cpp is set up, plus the command to start it.""" + checkout = _build.find_local_checkout() + details: List[str] = [] + launch = "" + if checkout is None: + # No local checkout: only a remote server can make this usable. + remote = _detect_remote() + return BackendStatus("audiocpp", "audio.cpp", installed=False, + configured=False, running=remote[0], + remote=remote[0], remote_urls=remote[1], + details=["not cloned — run setup to clone " + "./app/audio.cpp"]) + details.append(f"checkout: {checkout}") + binary = _build.find_audiocpp_server_bin(checkout) + built = binary is not None + if built: + details.append(f"built: {binary}") + else: + details.append("not built — run setup to build audiocpp_server") + server_json = checkout / "server.json" + configured = server_json.exists() + specs: List[ServerSpec] = [] + missing = _models.missing_model_entries(server_json) if configured else [] + if configured: + details.append(f"config: {server_json}") + if missing: + # The config references model files that are not on disk; a + # conversion would fail at model-load time, so say so now. + details.extend(_models.model_install_hints(checkout, missing)) + if built: + # Spawned from the checkout: audiocpp_server discovers + # model_specs/<family>.json relative to its working directory. + specs = [ServerSpec( + "audiocpp", config.AUDIOCPP_API_URL, + [str(binary), "--config", str(server_json)], + cwd=checkout, identity=probe.IDENTITY_AUDIOCPP)] + else: + launch = (f"cd {checkout} && ./build/<platform>-<backend>-release" + f"/bin/audiocpp_server --config {server_json}") + else: + details.append("no server.json — run setup to configure models") + if specs: + launch = format_launch_hint(specs) + managed = servers.manages(specs) + remote_running, remote_urls = _detect_remote(managed) + # A more specific "part-way set up" label than unavailable/installed: + # cloned but never built, or built but not configured. + partial = "" + if not built: + partial = "downloaded (not built)" + elif not configured: + partial = "built (not configured)" + return BackendStatus("audiocpp", "audio.cpp", installed=built, + configured=configured, + running=managed or remote_running, + details=details, launch_hint=launch, + servers=specs, managed=managed, + remote=remote_running, remote_urls=remote_urls, + models_missing=bool(missing), partial=partial) + + +def _detect_remote(managed: bool = False) -> Tuple[bool, dict]: + """Detect an externally-run audiocpp_server at the remote URL. + + Returns ``(running, {spec_name: url})``. The remote URL is probed only + when configured (non-empty); a server answering there is ignored when it + is this tool's own managed server (remote URL == local URL and our pid is + still alive) — that instance is already reported as "[local]". + """ + url = (config.AUDIOCPP_REMOTE_URL or "").strip() + if not url: + return False, {} + if managed and probe.same_endpoint(url, config.AUDIOCPP_API_URL): + return False, {} + if probe.identify_server(url) == probe.IDENTITY_AUDIOCPP: + return True, {"audiocpp": url} + return False, {} + + diff --git a/app/backends/audiocpp/voices.py b/app/backends/audiocpp/voices.py new file mode 100644 index 0000000..2f0fdd7 --- /dev/null +++ b/app/backends/audiocpp/voices.py @@ -0,0 +1,146 @@ +"""Reference-.wav transcription planning and execution.""" + +import argparse +from pathlib import Path +from typing import Callable, Dict, List, Optional, Tuple + +from backends.common import (PROMPT_TEXT_FILENAME, find_wav_files, + read_prompt_text) +from converter.clients import (transcribe_reference_audio, + whisper_backend_available) + +def transcribe_wav_dir(wav_files: list, whisper_model: str, + cancel=None) -> Dict[str, str]: + """Transcribe each wav file and return a mapping of stem -> transcript. + + CANCEL (a ``threading.Event``) is checked between files so the in-TUI + task view can stop a long transcription early. + """ + transcripts: Dict[str, str] = {} + for wav_file in wav_files: + if cancel is not None and cancel.is_set(): + print("[INFO] Transcription cancelled") + break + name = wav_file.stem + print(f"[INFO] Transcribing {wav_file.name} (voice '{name}')...") + text = transcribe_reference_audio(str(wav_file), model_name=whisper_model) + if text: + print(f"[OK] {name}: {text}") + else: + print(f"[WARNING] No transcript for '{name}'; cloning works best " + "with an accurate transcript — consider editing prompt_text " + "by hand before starting the server") + transcripts[name] = text or "" + return transcripts + + +def print_empty_transcript_warning(transcripts: Dict[str, str]) -> None: + """Print a loud, final warning for voices whose transcript is empty.""" + empty = sorted(name for name, text in transcripts.items() if not text) + if not empty: + return + bar = "=" * 70 + print() + print(bar) + print("[WARNING] MANUAL TRANSCRIPTION REQUIRED") + print(bar) + listing = " - " + "\n - ".join(empty) if len(empty) > 1 else f" - {empty[0]}" + print(f"The following voice(s) have an EMPTY transcript in prompt_text:\n" + f"{listing}") + print("Those voices will NOT work until you add an accurate transcript.") + print(f"Edit {PROMPT_TEXT_FILENAME} in your voice directory and fill in the " + "text after '|' for each voice above.") + print(bar) + + +def _decide_transcription(wav_files: list, existing: Dict[str, str], + prompt_exists: bool, force: bool, + confirm: Callable[[str, bool], bool]) -> dict: + """Decide which voices to transcribe; CONFIRM asks the plan questions. + + Returns a plan dict: {"mode": "all"|"missing"|"keep", "missing": + [...], "existing": {...}} — "existing" carries the prompt_text + mapping read while deciding, so the caller can reuse it instead of + reading the file again. + """ + mode = "all" + missing: List[Path] = [] + if prompt_exists and not force: + missing = [wav for wav in wav_files + if not existing.get(wav.stem, "").strip()] + if not missing: + if confirm("All voices already transcribed in prompt_text. " + "Re-transcribe anyway?", False): + mode = "all" + else: + mode = "keep" + elif confirm("Existing transcription and new .wavs detected, " + "only transcribe new voices?", True): + mode = "missing" + else: + mode = "all" + return {"mode": mode, "missing": missing, "existing": existing} + + +def _transcribe(args: argparse.Namespace, plan: Optional[dict], + cancel=None) -> Tuple[Dict[str, str], bool]: + """Transcribe the wav directory into a stem -> transcript mapping. + + Returns the mapping and a flag indicating whether it should be written to + prompt_text (False when an existing, complete prompt_text is kept as-is). + PLAN is always pre-collected — by the TUI (via _decide_transcription and + its confirm callbacks) or by _flag_plan for a non-interactive run — so no + questions are asked here; a None PLAN defaults to "transcribe everything". + CANCEL is checked between files. + """ + wav_files = find_wav_files(args.input_dir) + if not wav_files: + print(f"[WARNING] No .wav files found in {args.input_dir}; writing the " + "config without a voice_dir") + return {}, False + + prompt_path = args.input_dir / PROMPT_TEXT_FILENAME + existing = dict((plan or {}).get("existing") or {}) + mode = plan["mode"] if plan else "all" + + if mode == "keep": + print(f"[INFO] Kept existing {prompt_path}; all voices were " + "already transcribed, nothing new to transcribe") + return existing, False + + if whisper_backend_available() is None: + print("[WARNING] Neither faster_whisper nor whisper was found, so " + "reference .wav files cannot be transcribed automatically and " + "every transcript will be empty.") + print(" Install whisper (or faster_whisper) in your " + "audiobook environment to transcribe automatically; otherwise " + "transcripts must be added by hand (see the warning at the end).") + + if plan["mode"] == "missing": + new_transcripts = transcribe_wav_dir(plan["missing"], args.whisper_model, + cancel=cancel) + transcripts = dict(existing) + transcripts.update(new_transcripts) + else: + transcripts = transcribe_wav_dir(wav_files, args.whisper_model, + cancel=cancel) + return transcripts, True + + +def _flag_plan(wav_files: list, prompt_path: Path, force: bool) -> dict: + """Build a transcription plan for a non-interactive (flag-only) run. + + With --force everything is re-transcribed; otherwise an existing + prompt_text is reused and only voices with an empty transcript are + re-transcribed, mirroring what the TUI confirms interactively. + """ + if prompt_path.exists() and not force: + existing = read_prompt_text(prompt_path) + missing = [wav for wav in wav_files + if not existing.get(wav.stem, "").strip()] + if not missing: + return {"mode": "keep", "missing": [], "existing": existing} + return {"mode": "missing", "missing": missing, "existing": existing} + return {"mode": "all", "missing": [], "existing": {}} + + diff --git a/app/backends/audiocpp/wizard.py b/app/backends/audiocpp/wizard.py new file mode 100644 index 0000000..dcab273 --- /dev/null +++ b/app/backends/audiocpp/wizard.py @@ -0,0 +1,1081 @@ +"""The audio.cpp setup wizard: TUI screens, task lanes, CLI entry points.""" + +import argparse +import json +import sys +from pathlib import Path +from typing import Callable, Dict, List, Optional, Tuple + +from backends import common +from backends.common import ( + APP_DIR, + PROMPT_TEXT_FILENAME, + TTS_ROOT, + VOICES_DIR, + detect_wav_dir, + find_wav_files, + read_prompt_text, + resolve_wav_dir_arg, + wav_dir_info as _wav_dir_info, + wav_dir_preview as _wav_dir_preview, + write_prompt_text, +) +from converter import config +from ui import taskview, tui +from . import build as _build +from . import configsync as _configsync +from . import models as _models +from . import voices as _voices +from .catalog import (BACKENDS, DEFAULT_HOST, _backend_options, + build_model_entry, build_server_config, detect_backend, + load_model_catalog, load_server_config, + package_dir_options, server_config_selections) +from .constants import (AUDIOCPP_DIR_NAME, AUDIOCPP_GIT_URL, + TASK_TTS, TASK_VDES) + +_GO_BACK = object() + + +class _GoBack(Exception): + """Internal signal: Esc was pressed inside one of a screen's sub-prompts. + + The wizard drives a stack of screens via ``tui.Wizard``. Helpers that ask + several questions through callbacks (the task/id pickers inside + ``_build_entries``, the transcription plan, the download prompt) cannot + themselves return the wizard's ``BACK`` sentinel, so they convert the + ``_GO_BACK`` value passed to each widget into this exception. The screen + that invoked the helper catches it and returns ``tui.Wizard.BACK``, which + pops back to the previous screen. Esc on the first screen aborts the + whole wizard. + """ + + +class _TuiError(Exception): + """A fatal error raised from inside the TUI wizard. + + The message is reported to stderr after the terminal is restored; the + process exits with code 2 (matching a parser error). + """ + + +# Alias kept on this module: main()'s tty check and its tests patch it +# here. +from backends.setup import interactive as _interactive + + +def _build_entries(family_keys: List[str], chosen: Dict[str, List[dict]], + catalog_by_family: Dict[str, dict], + task_picker: Callable[[str], str], + known_tasks: Optional[Dict[Tuple[str, str], str]] = None + ) -> Tuple[List[dict], List[str], List[Tuple[str, str]], + List[str], bool]: + """Build server.json model entries from the selected families/packages. + + TASK_PICKER is called for each design package to choose vdes/tts. + KNOWN_TASKS maps ``(family, target_directory)`` to a previously-stored + task ("tts" or "vdes") so a modify run preserves how a design package + was hosted instead of re-asking. Each entry's server id is its package + ``target_directory`` (flattened to a token), so packages from the same + family never collide; an id that does collide (across families) is + auto-suffixed without prompting. Returns (model_entries, entry_ids, + install_guidance, design_entry_ids, include_clone). + """ + model_entries: List[dict] = [] + entry_ids: List[str] = [] + install_guidance: List[Tuple[str, str]] = [] + design_entry_ids: List[str] = [] + include_clone = False + for family in family_keys: + entry = catalog_by_family[family] + include_clone = include_clone or entry["clone_capable"] + for opt in chosen[family]: + if opt["design"]: + task = known_tasks.get((family, opt["target_directory"])) \ + if known_tasks else None + if task is None: + task = task_picker(opt["install_id"]) + else: + task = TASK_TTS + base_id = opt["target_directory"].replace("/", "-") + model_id = base_id + if model_id in entry_ids: + n = 2 + while f"{base_id}-{n}" in entry_ids: + n += 1 + model_id = f"{base_id}-{n}" + entry_ids.append(model_id) + model_entries.append(build_model_entry( + family, model_id, f"models/{opt['target_directory']}", + task=task)) + install_guidance.append((entry["display_name"], opt["install_id"])) + if task == TASK_VDES: + design_entry_ids.append(model_id) + return (model_entries, entry_ids, install_guidance, + design_entry_ids, include_clone) + + +def _write_and_advise(audiocpp_dir: Path, wav_dir: Optional[Path], + output_path: Path, model_entries: List[dict], + install_guidance: List[Tuple[str, str]], host: str, + port: int, backend: str, lazy_load: bool, + transcripts: Dict[str, str], write_prompt: bool) -> None: + """Console phase shared by both UI modes: write files, print summary. + + After a successful run the console output is the path of the written + server.json. The model install commands (and optional automatic + download) are handled separately by _install_models, called by both + UI modes once the user has decided whether to download. + """ + voice_dir: Optional[str] = None + if transcripts: + if write_prompt: + prompt_path = wav_dir / PROMPT_TEXT_FILENAME + write_prompt_text(wav_dir, transcripts) + print(f"[OK] Wrote {prompt_path}") + voice_dir = str(wav_dir.resolve()) + + server_config = build_server_config( + host=host, port=port, backend=backend, lazy_load=lazy_load, + model_entries=model_entries, voice_dir=voice_dir) + + with output_path.open("w", encoding="utf-8") as handle: + json.dump(server_config, handle, indent=2, ensure_ascii=False) + handle.write("\n") + + count = len(model_entries) + print(f"Wrote {output_path.resolve()} with {count} " + f"{'entry' if count == 1 else 'entries'}.") + + +def _wizard(stdscr, args: argparse.Namespace, parser: argparse.ArgumentParser + ) -> Optional[dict]: + """Run every TUI screen; return the collected settings, or None to abort. + + The wizard is driven by ``tui.Wizard`` as a stack of screen closures: + each screen shows one interactive widget and returns the next screen + (a closure), ``Wizard.BACK`` (Esc/q pressed — pop to the previous + screen), or the final settings dict. Only screens that actually render + are pushed, so Esc always lands on the previous real screen. A step + whose value is already provided by a flag (``--host``, ``--port``, + ``--families``, ...) or does not apply (e.g. the port-sync prompt when + the port did not change) is folded into the ``_after_*`` guards and + never becomes a screen. Esc on the first screen aborts the whole + wizard. + """ + + s: dict = {} + + def ask_confirm(question: str, default: bool) -> bool: + result = tui.confirm(stdscr, question, default=default, + cancel_value=_GO_BACK) + if result is _GO_BACK: + raise _GoBack() + return result + + def resolve_checkout(audiocpp_dir: Path) -> None: + """Validate the audio.cpp checkout and populate the wizard state ``s``.""" + audiocpp_dir = Path(audiocpp_dir).resolve() + try: + catalog = load_model_catalog(audiocpp_dir) + except NotADirectoryError as exc: + raise _TuiError(str(exc)) + if not catalog: + raise _TuiError(f"No TTS model families found in " + f"{audiocpp_dir}/model_specs; check the " + "checkout is up to date") + catalog_by_family = {entry["family"]: entry for entry in catalog} + output_path = args.output if args.output is not None \ + else audiocpp_dir / "server.json" + # Modify flow: an existing server.json seeds the wizard's screens + # instead of being overwritten from scratch (an explicit --force + # still starts fresh). + existing_config = load_server_config(output_path) \ + if not args.force else None + if existing_config is not None: + existing_selected, existing_tasks = \ + server_config_selections(existing_config, catalog) + else: + existing_selected, existing_tasks = {}, {} + s.update({ + "audiocpp_dir": audiocpp_dir, + "catalog": catalog, + "catalog_by_family": catalog_by_family, + "output_path": output_path, + "existing_config": existing_config, + "existing_selected": existing_selected, + "existing_tasks": existing_tasks, + "existing_host": existing_config.get("host") + if existing_config else None, + "existing_port": existing_config.get("port") + if existing_config else None, + "existing_backend": existing_config.get("backend") + if existing_config else None, + "existing_voice_dir": existing_config.get("voice_dir") + if existing_config else None, + "detected_backend": detect_backend(audiocpp_dir), + }) + + def _families_from_flag() -> None: + requested = [f.strip() for f in args.families.split(",") if f.strip()] + unknown = [f for f in requested if f not in s["catalog_by_family"]] + if unknown: + raise _TuiError( + f"Unknown family in --families: {', '.join(unknown)}. " + f"Available: {', '.join(s['catalog_by_family'])}") + chosen: Dict[str, List[dict]] = {} + family_keys: List[str] = [] + for family in requested: + if family not in family_keys: + family_keys.append(family) + chosen[family] = [opt for opt in package_dir_options( + s["catalog_by_family"][family]) if opt["recommended"]] + s["chosen"] = chosen + s["family_keys"] = family_keys + + def _compute_entries() -> None: + # Design task menu. Esc raises _GoBack, which the caller turns into + # Wizard.BACK (the design prompts are grouped: Esc returns to the + # families tree). + def task_picker(install_id: str) -> str: + result = tui.menu( + stdscr, + f"How should the '{install_id}' package be hosted?", + [ + ("design (vdes) - describe the voice with " + "--instructions", TASK_VDES), + ("tts - normal synthesis", TASK_TTS), + ], default_index=0, back_value=_GO_BACK) + if result is _GO_BACK: + raise _GoBack() + return result + + model_entries, entry_ids, install_guidance, \ + design_entry_ids, include_clone = _build_entries( + s["family_keys"], s["chosen"], s["catalog_by_family"], + task_picker, known_tasks=s["existing_tasks"]) + s.update({ + "model_entries": model_entries, + "entry_ids": entry_ids, + "install_guidance": install_guidance, + "design_entry_ids": design_entry_ids, + "include_clone": include_clone, + }) + + def _finalize() -> dict: + return { + "audiocpp_dir": s["audiocpp_dir"], + "catalog": s["catalog"], + "catalog_by_family": s["catalog_by_family"], + "output_path": s["output_path"], + "family_keys": s["family_keys"], + "chosen": s["chosen"], + "model_entries": s["model_entries"], + "entry_ids": s["entry_ids"], + "install_guidance": s["install_guidance"], + "design_entry_ids": s["design_entry_ids"], + "include_clone": s["include_clone"], + "host": s["host"], + "port": s["port"], + "backend": s["backend"], + "build": s["build"], + "lazy_load": s["lazy_load"], + "sync_port": s["sync_port"], + "sync_model_ids": s["sync_model_ids"], + "wav_dir": s["wav_dir"], + "plan": s["plan"], + "download": s["download"], + "delete_unused": s["delete_unused"], + "unused_entries": s["unused_entries"], + } + + def screen_families(): + """Pick TTS model families and packages (the modify tree).""" + tree_families = _models._build_tree_families(s["catalog"]) + # Modify flow: pre-check the models an existing server.json hosts, + # so the tree opens as a "modify" list rather than a fresh one. + checked_set = set() + for family, dirs in s["existing_selected"].items(): + if family not in s["catalog_by_family"]: + continue + family_index = s["catalog"].index(s["catalog_by_family"][family]) + valid_dirs = {opt["target_directory"] + for opt in package_dir_options( + s["catalog_by_family"][family])} + for target in dirs: + if target in valid_dirs: + checked_set.add((family_index, target)) + picked = tui.checkbox_tree( + stdscr, "Select TTS model families to host", + tree_families, expand_all=args.all_packages, + back_value=_GO_BACK, checked=checked_set) + if picked is _GO_BACK: + return tui.Wizard.BACK + chosen: Dict[str, List[dict]] = {} + family_keys: List[str] = [] + for family_index, option_key in picked: + family = s["catalog"][family_index]["family"] + if family not in chosen: + chosen[family] = [] + family_keys.append(family) + chosen[family].append(option_key) + for family in list(chosen): + keyed = {opt["target_directory"]: opt + for opt in package_dir_options( + s["catalog_by_family"][family])} + chosen[family] = [keyed[key] for key in chosen[family]] + s["chosen"] = chosen + s["family_keys"] = family_keys + return screen_host + + def _after_families(): + if args.families is not None: + _families_from_flag() + return screen_host + return screen_families + + def screen_host(): + """Build the model entries, then ask the bind host. + + The task/id pickers (when any) run here too and are grouped with + this screen: Esc on one of them (or on the host field) returns to + the families tree. + """ + try: + _compute_entries() + except _GoBack: + return tui.Wizard.BACK + if args.host is not None: + s["host"] = args.host + return _after_host() + host = tui.line_edit( + stdscr, "Bind host", + s["existing_host"] if isinstance(s["existing_host"], str) + else DEFAULT_HOST, + help_lines=["The IP address audiocpp will be hosted on", + "127.0.0.1 (this machine) is probably " + "correct"], back_value=_GO_BACK) + if host is _GO_BACK: + return tui.Wizard.BACK + s["host"] = host + return _after_host() + + def _after_host(): + if args.port is None: + return screen_port + s["port"] = args.port + return _after_port() + + def screen_port(): + port_text = tui.line_edit( + stdscr, "Port", + str(s["existing_port"]) if isinstance(s["existing_port"], int) + else str(_configsync.config_port()), + validate=lambda s: None if (s.isdigit() + and 1 <= int(s) <= 65535) + else "Enter a port number between 1 and 65535", + help_lines=["The port audiocpp will be hosted on"], + back_value=_GO_BACK) + if port_text is _GO_BACK: + return tui.Wizard.BACK + s["port"] = int(port_text) + return _after_port() + + def _after_port(): + s["sync_port"] = None + if s["port"] != _configsync.config_port(): + return screen_sync_port + return _after_sync() + + def screen_sync_port(): + sync_port = tui.confirm( + stdscr, "Update AUDIOCPP_API_URL in app/converter/config.py " + f"to port {s['port']} so audiobook.py talks to this server", + default=True, cancel_value=_GO_BACK) + if sync_port is _GO_BACK: + return tui.Wizard.BACK + s["sync_port"] = sync_port + return _after_sync() + + def _after_sync(): + if args.build_backend: + s["backend"] = args.build_backend + s["build"] = s["detected_backend"] is None + return _after_backend() + if args.backend: + s["backend"] = args.backend + s["build"] = False + return _after_backend() + if s["detected_backend"] is not None: + # Already built: use the detected backend, no menu, no build. + s["backend"] = s["detected_backend"] + s["build"] = False + return _after_backend() + # Not built for any backend yet: always ask which backend the server + # should use and offer to build it — even on a modify run, so a user + # who declined the build the first time is never stranded without a + # way to build from the TUI. + return screen_backend + + def screen_backend(): + # Pre-select the backend an existing server.json records (modify + # flow), so re-running setup lands on the previous choice. + backend_options, backend_default = _backend_options(None) + if s["existing_backend"] in BACKENDS: + backend_default = next( + (index for index, (_label, value) in enumerate(backend_options) + if value == s["existing_backend"]), backend_default) + backend = tui.menu( + stdscr, "Which inference backend should audiocpp_server " + "use?", backend_options, + default_index=backend_default, back_value=_GO_BACK) + if backend is _GO_BACK: + return tui.Wizard.BACK + s["backend"] = backend + if _build.built_server_binary(s["audiocpp_dir"], backend) is not None: + # A checkout with builds for several backends: this one is + # already built, so there is nothing to build. + s["build"] = False + return _after_backend() + return screen_build + + def screen_build(): + # Not built for the chosen backend yet: offer to build it now. The + # build itself runs in the TUI task view (or the console tail for + # CLI runs) after the wizard. + build = tui.confirm( + stdscr, f"audiocpp_server is not built for {s['backend']}. " + f"Build it now (runs scripts/build_*)?", + default=True, cancel_value=_GO_BACK) + if build is _GO_BACK: + return tui.Wizard.BACK + s["build"] = build + return _after_backend() + + def _after_backend(): + s["lazy_load"] = True + return _after_lazy() + + def _after_lazy(): + if args.input_dir is not None: + s["wav_dir"] = args.input_dir + return _after_wav() + if s["include_clone"]: + return screen_wav + s["wav_dir"] = None + return _after_wav() + + def screen_wav(): + wav_start = detect_wav_dir(s["audiocpp_dir"], TTS_ROOT) + # Modify flow: an existing voice_dir seeds the browser so the user + # can accept it on Enter instead of re-navigating. + if isinstance(s["existing_voice_dir"], str) and s["existing_voice_dir"]: + wav_start = Path(s["existing_voice_dir"]) + wav_dir = tui.browse_directory( + stdscr, "Select the directory with your .wav voices", + info=_wav_dir_info, preview=_wav_dir_preview, + start=wav_start if wav_start is not None else VOICES_DIR, + back_value=_GO_BACK) + if wav_dir is _GO_BACK: + return tui.Wizard.BACK + s["wav_dir"] = wav_dir + return _after_wav() + + def _after_wav(): + s["plan"] = None + if s["include_clone"] and s["wav_dir"] is not None: + wav_files = find_wav_files(s["wav_dir"]) + if wav_files: + prompt_path = s["wav_dir"] / PROMPT_TEXT_FILENAME + if prompt_path.exists() and not args.force: + return screen_transcription + existing = read_prompt_text(prompt_path) if ( + prompt_path.exists() and not args.force) else {} + s["plan"] = _voices._decide_transcription( + wav_files, existing, prompt_path.exists(), + args.force, ask_confirm) + return _after_transcription() + + def screen_transcription(): + # Transcription plan (questions only; transcription runs after). + wav_files = find_wav_files(s["wav_dir"]) + prompt_path = s["wav_dir"] / PROMPT_TEXT_FILENAME + existing = read_prompt_text(prompt_path) if ( + prompt_path.exists() and not args.force) else {} + try: + s["plan"] = _voices._decide_transcription( + wav_files, existing, prompt_path.exists(), + args.force, ask_confirm) + except _GoBack: + return tui.Wizard.BACK + return _after_transcription() + + def _after_transcription(): + s["sync_model_ids"] = None + if len(s["entry_ids"]) == 1 and not ( + config.AUDIOCPP_MODEL_ID == s["entry_ids"][0] + and config.AUDIOCPP_CLONE_MODEL_ID == s["entry_ids"][0]): + return screen_model_sync + return _after_model_sync() + + def screen_model_sync(): + sync_model_ids = tui.confirm( + stdscr, "Update AUDIOCPP_MODEL_ID and " + "AUDIOCPP_CLONE_MODEL_ID in app/converter/config.py to " + f"'{s['entry_ids'][0]}' so audiobook.py uses this model", + default=True, cancel_value=_GO_BACK) + if sync_model_ids is _GO_BACK: + return tui.Wizard.BACK + s["sync_model_ids"] = sync_model_ids + return _after_model_sync() + + def _after_model_sync(): + new_paths = {entry["path"] for entry in s["model_entries"]} + s["unused_entries"] = _models.unused_installed_entries( + s["output_path"], new_paths) \ + if s["existing_config"] is not None else [] + s["delete_unused"] = False + if s["unused_entries"]: + return screen_delete_unused + return _after_delete() + + def screen_delete_unused(): + delete_unused = tui.confirm( + stdscr, "Delete unused models?", default=False, + cancel_value=_GO_BACK) + if delete_unused is _GO_BACK: + return tui.Wizard.BACK + s["delete_unused"] = delete_unused + return _after_delete() + + def _after_delete(): + manager = s["audiocpp_dir"] / "tools" / "model_manager_v2.py" + if manager.is_file(): + return screen_download + s["download"] = False + return _finalize() + + def screen_download(): + # Automatic model download (or print the install commands). + try: + s["download"] = _models._decide_download( + s["audiocpp_dir"], s["model_entries"], ask_confirm) + except _GoBack: + return tui.Wizard.BACK + return _finalize() + + # First screen: resolve the checkout directly when it already exists + # (the modify flow), so the wizard starts on a real screen. When no + # checkout exists, clone it into ./app/audio.cpp (streaming inside the + # TUI task view, not by dropping to the console) without asking, then + # continue the same way. + audiocpp_dir = _build.find_local_checkout() + if audiocpp_dir is None: + target = APP_DIR / AUDIOCPP_DIR_NAME + rc = taskview.run_steps(stdscr, "Clone audio.cpp", [ + taskview.TaskStep( + f"Cloning audio.cpp into {target}", + lambda emit, cancel: common.git_clone( + AUDIOCPP_GIT_URL, target, emit=emit, cancel=cancel)), + taskview.TaskStep( + "Apply ggml build patches", + lambda emit, cancel: _build.apply_ggml_patches( + target, emit=emit, cancel=cancel)), + ]) + if rc == 130: + # Cancelled from the task view: abort the wizard quietly. + return None + if rc != 0: + raise _TuiError( + f"audio.cpp setup step failed (exit {rc}). Clone " + f"audio.cpp manually: git clone " + f"{AUDIOCPP_GIT_URL} {target}, then re-run") + audiocpp_dir = target + resolve_checkout(audiocpp_dir) + first = _after_families() + return tui.Wizard().run(first) + + +def _execute_lanes(settings: dict, + args: argparse.Namespace) -> List[taskview.TaskLane]: + """Build the ordered setup steps for the in-TUI task view, per lane. + + The same work ``_execute`` runs on the console, split into two lanes so + the view can run the build in one pane while configuring and downloading + models in the other (both progress bars visible at once). The build lane + exists only when ``settings["build"]`` is set; the models lane always + exists (transcribe → write server.json → download/print commands). + Shared results (the transcription mapping) travel through a small closure + dict scoped to the models lane. Each step's ``work(emit, cancel)`` + returns its exit code; subprocess steps stream through EMIT and abort on + CANCEL, while print()-based steps are captured by the view's stdout + routing. + """ + audiocpp_dir = settings["audiocpp_dir"] + state: dict = {} + build = settings.get("build") + lanes: List[taskview.TaskLane] = [] + + if build: + def build_step(emit, cancel): + rc = _build.build_audiocpp(audiocpp_dir, settings["backend"], + emit=emit, cancel=cancel) + if rc != 0: + print(f"[WARNING] build exited with code {rc}; the server.json " + "was still written — build audiocpp_server manually " + "before starting it") + else: + print("[OK] build complete") + return rc + lanes.append(taskview.TaskLane( + "Build", + [taskview.TaskStep( + f"Build audiocpp_server ({settings['backend']})", + build_step)])) + + def transcribe(emit, cancel): + args.input_dir = settings["wav_dir"] + if settings["include_clone"] and args.input_dir is not None: + transcripts, write_prompt = _voices._transcribe( + args, plan=settings["plan"], cancel=cancel) + elif args.input_dir is not None: + print(f"[WARNING] Ignoring {args.input_dir}: no clone-capable " + "family selected, so voice presets are not used") + transcripts, write_prompt = {}, False + else: + transcripts, write_prompt = {}, False + state["transcripts"] = transcripts + state["write_prompt"] = write_prompt + return 0 + + def write(emit, cancel): + # Port sync (applied now that the terminal is back). + if settings["sync_port"] is True: + _configsync._apply_port_sync(settings["port"], True) + elif settings["sync_port"] is False: + _configsync._apply_port_sync(settings["port"], False) + + _write_and_advise( + audiocpp_dir, settings["wav_dir"], settings["output_path"], + settings["model_entries"], settings["install_guidance"], + settings["host"], settings["port"], settings["backend"], + settings["lazy_load"], state["transcripts"], state["write_prompt"]) + + # Delete-unused cleanup (modify flow): remove the already-downloaded + # models the new selection dropped. The regenerated server.json + # already only lists the kept entries. + if settings.get("delete_unused"): + removed = _models.delete_model_files(settings["output_path"], + settings["unused_entries"]) + print(f"[OK] Deleted {removed} unused model " + f"{'entry' if removed == 1 else 'entries'} from disk.") + + if len(settings["entry_ids"]) == 1: + _configsync._offer_config_model_id_sync(settings["entry_ids"][0], + settings["sync_model_ids"]) + _voices.print_empty_transcript_warning(state["transcripts"]) + return 0 + + def install(emit, cancel): + _models._install_models(audiocpp_dir, settings["install_guidance"], + settings["download"], emit=emit, cancel=cancel) + _build._print_launch_hint(audiocpp_dir, settings["output_path"]) + return 0 + install_title = "Download models" if settings.get("download") \ + else "Print model install commands" + + lanes.append(taskview.TaskLane( + "Configure & download", + [taskview.TaskStep("Transcribe reference voices", transcribe), + taskview.TaskStep("Write server.json & sync config", write), + taskview.TaskStep(install_title, install)])) + + return lanes + + +def _execute_steps(settings: dict, + args: argparse.Namespace) -> List[taskview.TaskStep]: + """The ordered setup steps for the sequential console path. + + The lanes ``_execute_lanes`` builds, flattened into one ordered list + (build first, then transcribe → write → download), so the console tail + is byte-identical to the pre-lanes behavior. + """ + steps: List[taskview.TaskStep] = [] + for lane in _execute_lanes(settings, args): + steps.extend(lane.steps) + return steps + + +def _execute(settings: dict, args: argparse.Namespace) -> int: + """Shared console tail: build, sync, transcribe, write, install, advise. + + Runs after the TUI wizard returns (or after _collect_from_flags for a + non-interactive run): the terminal is plain, so subprocess output and + transcription progress appear normally. The same work as + ``_execute_steps``, run with no emit (console streaming). + """ + return taskview.run_steps_inline(_execute_steps(settings, args)) + + +def setup_screen(stdscr) -> int: + """Run the setup wizard on an existing curses screen (the hub's). + + The hub drives this as one screen of its own ``tui.Wizard`` stack, so + Esc on the wizard's first screen simply returns here and the hub pops + back to the menu that launched it. The setup tail (build, transcribe, + write, download) runs inside the TUI task view on this same screen, so + the hub's curses session stays intact and the user sees per-step status + and progress instead of being dropped to the console. On a fresh install + the build and the model setup run as two parallel lanes (a split view), + so cloning → configuring → building+downloading is one continuous, + one-click flow; the individual "Build" and "Download Missing Models" hub + actions remain only as fallbacks when something fails or is interrupted. + Returns 0 on completion, 1 when the user aborted. + """ + parser = build_parser() + args = parser.parse_args([]) + settings = _wizard(stdscr, args, parser) + if settings is None: + return 1 + return taskview.run_lanes(stdscr, "Setting up audio.cpp", + _execute_lanes(settings, args)) + + +def build_screen(stdscr) -> int: + """Build audiocpp_server from the hub when the checkout has no binary. + + Asks which backend to build for (pre-selecting the backend an existing + server.json records, else cuda), runs the build inside the TUI task view + — alongside a download of any missing models when server.json is already + configured and those models map to an install command (the split view), + or just the build otherwise — then updates server.json's ``backend`` + field to match. Returns 0 on success, non-zero when the user backed out, + cancelled, or the build failed. This is the hub's "Build audio.cpp + server" action, so a checkout that was cloned but never built is always + buildable from the TUI; the standalone "Download Missing Models" action + stays as the fallback when the download fails or is interrupted. + """ + checkout = _build.find_local_checkout() + if checkout is None: + tui.flash(stdscr, "No audio.cpp checkout found — install audio.cpp " + "first.", "err") + return 1 + if _build.find_audiocpp_server_bin(checkout) is not None: + tui.flash(stdscr, "audiocpp_server is already built.", "ok") + return 0 + server_config = load_server_config(checkout / "server.json") or {} + recorded = server_config.get("backend") + options, default = _backend_options(None) + if recorded in BACKENDS: + default = next((i for i, (_label, value) in enumerate(options) + if value == recorded), default) + backend = tui.menu( + stdscr, "Which inference backend should audiocpp_server be built " + "for?", options, default_index=default, back_value=_GO_BACK) + if backend is _GO_BACK: + return 1 + + def build_step(emit, cancel): + return _build.build_audiocpp(checkout, backend, emit=emit, cancel=cancel) + + lanes = [taskview.TaskLane( + "Build", [taskview.TaskStep( + f"Build audiocpp_server ({backend})", build_step)])] + + # Missing models this build can also fetch, so a configured backend that + # lost its binary is restored to "installed" in one step. + server_json = checkout / "server.json" + missing = _models.missing_model_entries(server_json) if server_json.exists() else [] + guidance = _models.missing_model_install_guidance(checkout, missing) \ + if missing else [] + + if guidance: + def download_step(emit, cancel): + _models.install_models(checkout, guidance, emit=emit, cancel=cancel) + return 0 + lanes.append(taskview.TaskLane( + "Download models", + [taskview.TaskStep("Download missing models", download_step)])) + + title = "Build & download models" if len(lanes) == 2 \ + else "Build audiocpp_server" + rc = taskview.run_lanes(stdscr, title, lanes) + if rc != 0: + return rc + if _configsync.update_server_backend(backend): + tui.flash(stdscr, f"audiocpp_server built for {backend}.", "ok") + else: + tui.flash(stdscr, f"audiocpp_server built for {backend}. (Could not " + "update server.json's backend field — reconfigure audio.cpp " + "if it was already configured.)", "warn") + # Models that can't be mapped to an install command still need hand + # installation; say so now rather than leaving the user in the dark. + if missing and not guidance: + tui.flash(stdscr, _models.hand_install_guidance(checkout, missing), "err") + return 0 + + +def run_tui(args: Optional[argparse.Namespace] = None, + parser: Optional[argparse.ArgumentParser] = None) -> int: + """Run the audio.cpp setup wizard end-to-end. + + With no ARGS (the hub's call) a default namespace is built so the full + wizard runs. Called from ``main`` after argparse when the terminal is + interactive. Returns the process exit code. + """ + import curses + if args is None: + parser = build_parser() + args = parser.parse_args([]) + if args.input_dir is not None and not args.input_dir.is_dir(): + print(f"[ERROR] --wavs not found: {args.input_dir}", + file=sys.stderr) + return 2 + try: + settings = curses.wrapper(_wizard, args, parser) + except _TuiError as exc: + print(f"[ERROR] {exc}", file=sys.stderr) + return 2 + except tui.WizardCancelled: + print("\n[INFO] Cancelled; nothing was written") + return 1 + try: + curses.curs_set(1) # restore the text cursor hidden by the TUI + except curses.error: + pass + if settings is None: + print("[INFO] Aborted; existing server.json kept") + return 1 + return _execute(settings, args) + + +def _collect_from_flags(args: argparse.Namespace, + parser: argparse.ArgumentParser) -> Optional[dict]: + """Build the settings dict from flags for a non-interactive run. + + Every required value must come from a flag (there are no prompts in a + non-interactive run); a missing one is a hard ``parser.error``. Returns + the settings dict, or None when the user declined an overwrite (the + default-location fallback then also exists). + """ + # Checkout: ./app/audio.cpp, else --clone clones one there. + audiocpp_dir = _build.find_local_checkout() + if audiocpp_dir is None and args.clone: + target = APP_DIR / AUDIOCPP_DIR_NAME + rc = common.git_clone(AUDIOCPP_GIT_URL, target) + if rc != 0: + parser.error(f"git clone failed (exit {rc}); clone audio.cpp " + f"manually: git clone {AUDIOCPP_GIT_URL} {target}") + patch_rc = _build.apply_ggml_patches(target) + if patch_rc != 0: + parser.error( + f"ggml build patches could not be applied to {target} " + f"(exit {patch_rc}); see messages above. The audio.cpp " + f"fork's vendored ggml may have changed — re-evaluate " + f"app/backends/patches/.") + audiocpp_dir = target + if audiocpp_dir is None: + parser.error( + "An audio.cpp checkout is required. Pass --clone to clone " + "app/audio.cpp, or run without flags for the TUI wizard.") + try: + catalog = load_model_catalog(audiocpp_dir) + except NotADirectoryError as exc: + parser.error(str(exc)) + if not catalog: + parser.error( + f"No TTS model families found in {audiocpp_dir}/model_specs; " + "check the checkout is up to date") + catalog_by_family = {entry["family"]: entry for entry in catalog} + + # Families: required from --families in a non-interactive run. + if args.families is None: + parser.error("--families is required in a non-interactive run (or run " + "without flags for the TUI wizard)") + requested = [f.strip() for f in args.families.split(",") if f.strip()] + unknown = [f for f in requested if f not in catalog_by_family] + if unknown: + parser.error( + f"Unknown family in --families: {', '.join(unknown)}. " + f"Available: {', '.join(catalog_by_family)}") + family_keys: List[str] = [] + for fam in requested: + if fam not in family_keys: + family_keys.append(fam) + + chosen: Dict[str, List[dict]] = {} + for family in family_keys: + opts = package_dir_options(catalog_by_family[family]) + if args.all_packages: + chosen[family] = opts + else: + chosen[family] = [opt for opt in opts if opt["recommended"]] + + # Non-interactive picker: design packages default to vdes. + def task_picker(install_id: str) -> str: + return TASK_VDES + + model_entries, entry_ids, install_guidance, design_entry_ids, include_clone = \ + _build_entries(family_keys, chosen, catalog_by_family, + task_picker) + + # Server settings. + host = args.host or DEFAULT_HOST + detected_backend = detect_backend(audiocpp_dir) + if args.build_backend: + backend = args.build_backend + build = detected_backend is None + elif args.backend: + backend = args.backend + build = False + elif detected_backend is not None: + backend = detected_backend + build = False + else: + backend = "cuda" + build = False + port = args.port if args.port is not None else _configsync.config_port() + lazy_load = True + + # Output path / overwrite (decline falls back to cwd, then aborts). + output_path = args.output if args.output is not None \ + else audiocpp_dir / "server.json" + if output_path.exists() and not args.force: + if args.output is None: + output_path = Path.cwd() / "server.json" + if output_path.exists() and not args.force: + print("[INFO] Aborted; existing server.json kept") + return None + else: + print("[INFO] Aborted; existing server.json kept") + return None + + # Config sync decisions (auto-apply unless explicitly declined). + sync_port: Optional[bool] = None + if port != _configsync.config_port(): + sync_port = not args.no_sync_port + sync_model_ids: Optional[bool] = None + if len(entry_ids) == 1 and not ( + config.AUDIOCPP_MODEL_ID == entry_ids[0] + and config.AUDIOCPP_CLONE_MODEL_ID == entry_ids[0]): + sync_model_ids = not args.no_sync_model_ids + + # Wav dir + transcription plan (defaults to the project's voices/ dir). + wav_dir = args.input_dir if args.input_dir is not None else VOICES_DIR + plan: Optional[dict] = None + if include_clone and wav_dir is not None: + wav_files = find_wav_files(wav_dir) + if wav_files: + prompt_path = wav_dir / PROMPT_TEXT_FILENAME + plan = _voices._flag_plan(wav_files, prompt_path, args.force) + + return { + "audiocpp_dir": audiocpp_dir, + "catalog": catalog, + "catalog_by_family": catalog_by_family, + "output_path": output_path, + "family_keys": family_keys, + "chosen": chosen, + "model_entries": model_entries, + "entry_ids": entry_ids, + "install_guidance": install_guidance, + "design_entry_ids": design_entry_ids, + "include_clone": include_clone, + "host": host, + "port": port, + "backend": backend, + "build": build, + "lazy_load": lazy_load, + "sync_port": sync_port, + "sync_model_ids": sync_model_ids, + "wav_dir": wav_dir, + "plan": plan, + "download": args.download, + } + + +def build_parser() -> argparse.ArgumentParser: + """The audio.cpp setup CLI (also used to build a default namespace).""" + parser = argparse.ArgumentParser( + description="Set up the audio.cpp TTS backend: clone/build, pick " + "models, write server.json, and sync app/converter/config.py.") + parser.add_argument("--wavs", type=resolve_wav_dir_arg, default=None, + dest="input_dir", metavar="WAV_DIR", + help="Directory with .wav reference files to publish as " + "a server-level voice_dir cloning library " + f"(default: {VOICES_DIR}; asked for when omitted " + "in the TUI)") + parser.add_argument("--output", type=Path, default=None, + help="Output path for server.json (default: " + "server.json inside the audio.cpp checkout; an " + "existing file is overwritten only with --force " + "or a TUI confirm)") + parser.add_argument("--clone", action="store_true", + help="Non-interactive: clone audio.cpp into " + "./app/audio.cpp when no checkout is found") + parser.add_argument("--families", type=str, default=None, + help="Comma-separated model families to host, as named " + "in the audio.cpp catalog (e.g. " + "qwen3_tts,higgs_audio_tts). Required in a " + "non-interactive run; skips the family tree in " + "the TUI") + parser.add_argument("--all-packages", action="store_true", + help="Host every installable package of each selected " + "family (distinct target_directory) instead of " + "only the recommended one. Voice-design packages " + "are hosted with task 'vdes'") + parser.add_argument("--host", type=str, default=None, + help="Bind host for the server (default: 127.0.0.1)") + parser.add_argument("--port", type=int, default=None, + help="Port for the server (default: the port in " + "AUDIOCPP_API_URL from app/converter/config.py)") + parser.add_argument("--backend", choices=BACKENDS, default=None, + help="Inference backend recorded in server.json " + "(default: auto-detected from the checkout's " + "build/ directory, else cuda)") + parser.add_argument("--build-backend", choices=BACKENDS, default=None, + help="Build audiocpp_server for this backend when it " + "is not built yet, and use it in server.json") + parser.add_argument("--whisper-model", type=str, default="base", + help="Whisper model size for transcription " + "(default: base)") + parser.add_argument("--force", action="store_true", + help="Overwrite the output file (and prompt_text) " + "without prompting; in the TUI, start the " + "wizard fresh instead of loading the existing " + "server.json") + parser.add_argument("--download", action="store_true", + help="Run model_manager_v2.py install for each hosted " + "model automatically (default: print the commands " + "only)") + parser.add_argument("--no-sync-port", action="store_true", + help="Do not rewrite AUDIOCPP_API_URL in " + "app/converter/config.py when --port differs") + parser.add_argument("--no-sync-model-ids", action="store_true", + help="Do not rewrite AUDIOCPP_MODEL_ID/" + "AUDIOCPP_CLONE_MODEL_ID for a single-entry server") + return parser + + +def main() -> int: + parser = build_parser() + args = parser.parse_args() + + if args.input_dir is not None and not args.input_dir.is_dir(): + parser.error( + f"WAV directory not found: {args.input_dir}\n" + f" (resolved from the current working directory: " + f"{Path.cwd()})\n" + " --wavs must be a directory containing the .wav " + "reference files to use as voice cloning presets") + + if _interactive(): + return run_tui(args, parser) + + # Non-interactive (no terminal, or all flags supplied): flag-only path. + settings = _collect_from_flags(args, parser) + if settings is None: + return 1 + return _execute(settings, args) + + |
