diff options
| author | historia <historiavg@proton.me> | 2026-08-26 02:25:55 -0400 |
|---|---|---|
| committer | historia <historiavg@proton.me> | 2026-08-26 02:25:55 -0400 |
| commit | 8b5c8697740ff415cf7f1d03c9fb5a8c8851d420 (patch) | |
| tree | 28c0323c54c896af5f89fb34b89a62e0fe0df291 /app/backends/audiocpp/wizard.py | |
| parent | acbd9ff2c91182d96c57ffb57bee6e9b3fcbcbd4 (diff) | |
| download | tts-audiobook-generator-8b5c8697740ff415cf7f1d03c9fb5a8c8851d420.tar.gz | |
refactor: audiocpp.py setup flow
Diffstat (limited to 'app/backends/audiocpp/wizard.py')
| -rw-r--r-- | app/backends/audiocpp/wizard.py | 1081 |
1 files changed, 1081 insertions, 0 deletions
diff --git a/app/backends/audiocpp/wizard.py b/app/backends/audiocpp/wizard.py new file mode 100644 index 0000000..dcab273 --- /dev/null +++ b/app/backends/audiocpp/wizard.py @@ -0,0 +1,1081 @@ +"""The audio.cpp setup wizard: TUI screens, task lanes, CLI entry points.""" + +import argparse +import json +import sys +from pathlib import Path +from typing import Callable, Dict, List, Optional, Tuple + +from backends import common +from backends.common import ( + APP_DIR, + PROMPT_TEXT_FILENAME, + TTS_ROOT, + VOICES_DIR, + detect_wav_dir, + find_wav_files, + read_prompt_text, + resolve_wav_dir_arg, + wav_dir_info as _wav_dir_info, + wav_dir_preview as _wav_dir_preview, + write_prompt_text, +) +from converter import config +from ui import taskview, tui +from . import build as _build +from . import configsync as _configsync +from . import models as _models +from . import voices as _voices +from .catalog import (BACKENDS, DEFAULT_HOST, _backend_options, + build_model_entry, build_server_config, detect_backend, + load_model_catalog, load_server_config, + package_dir_options, server_config_selections) +from .constants import (AUDIOCPP_DIR_NAME, AUDIOCPP_GIT_URL, + TASK_TTS, TASK_VDES) + +_GO_BACK = object() + + +class _GoBack(Exception): + """Internal signal: Esc was pressed inside one of a screen's sub-prompts. + + The wizard drives a stack of screens via ``tui.Wizard``. Helpers that ask + several questions through callbacks (the task/id pickers inside + ``_build_entries``, the transcription plan, the download prompt) cannot + themselves return the wizard's ``BACK`` sentinel, so they convert the + ``_GO_BACK`` value passed to each widget into this exception. The screen + that invoked the helper catches it and returns ``tui.Wizard.BACK``, which + pops back to the previous screen. Esc on the first screen aborts the + whole wizard. + """ + + +class _TuiError(Exception): + """A fatal error raised from inside the TUI wizard. + + The message is reported to stderr after the terminal is restored; the + process exits with code 2 (matching a parser error). + """ + + +# Alias kept on this module: main()'s tty check and its tests patch it +# here. +from backends.setup import interactive as _interactive + + +def _build_entries(family_keys: List[str], chosen: Dict[str, List[dict]], + catalog_by_family: Dict[str, dict], + task_picker: Callable[[str], str], + known_tasks: Optional[Dict[Tuple[str, str], str]] = None + ) -> Tuple[List[dict], List[str], List[Tuple[str, str]], + List[str], bool]: + """Build server.json model entries from the selected families/packages. + + TASK_PICKER is called for each design package to choose vdes/tts. + KNOWN_TASKS maps ``(family, target_directory)`` to a previously-stored + task ("tts" or "vdes") so a modify run preserves how a design package + was hosted instead of re-asking. Each entry's server id is its package + ``target_directory`` (flattened to a token), so packages from the same + family never collide; an id that does collide (across families) is + auto-suffixed without prompting. Returns (model_entries, entry_ids, + install_guidance, design_entry_ids, include_clone). + """ + model_entries: List[dict] = [] + entry_ids: List[str] = [] + install_guidance: List[Tuple[str, str]] = [] + design_entry_ids: List[str] = [] + include_clone = False + for family in family_keys: + entry = catalog_by_family[family] + include_clone = include_clone or entry["clone_capable"] + for opt in chosen[family]: + if opt["design"]: + task = known_tasks.get((family, opt["target_directory"])) \ + if known_tasks else None + if task is None: + task = task_picker(opt["install_id"]) + else: + task = TASK_TTS + base_id = opt["target_directory"].replace("/", "-") + model_id = base_id + if model_id in entry_ids: + n = 2 + while f"{base_id}-{n}" in entry_ids: + n += 1 + model_id = f"{base_id}-{n}" + entry_ids.append(model_id) + model_entries.append(build_model_entry( + family, model_id, f"models/{opt['target_directory']}", + task=task)) + install_guidance.append((entry["display_name"], opt["install_id"])) + if task == TASK_VDES: + design_entry_ids.append(model_id) + return (model_entries, entry_ids, install_guidance, + design_entry_ids, include_clone) + + +def _write_and_advise(audiocpp_dir: Path, wav_dir: Optional[Path], + output_path: Path, model_entries: List[dict], + install_guidance: List[Tuple[str, str]], host: str, + port: int, backend: str, lazy_load: bool, + transcripts: Dict[str, str], write_prompt: bool) -> None: + """Console phase shared by both UI modes: write files, print summary. + + After a successful run the console output is the path of the written + server.json. The model install commands (and optional automatic + download) are handled separately by _install_models, called by both + UI modes once the user has decided whether to download. + """ + voice_dir: Optional[str] = None + if transcripts: + if write_prompt: + prompt_path = wav_dir / PROMPT_TEXT_FILENAME + write_prompt_text(wav_dir, transcripts) + print(f"[OK] Wrote {prompt_path}") + voice_dir = str(wav_dir.resolve()) + + server_config = build_server_config( + host=host, port=port, backend=backend, lazy_load=lazy_load, + model_entries=model_entries, voice_dir=voice_dir) + + with output_path.open("w", encoding="utf-8") as handle: + json.dump(server_config, handle, indent=2, ensure_ascii=False) + handle.write("\n") + + count = len(model_entries) + print(f"Wrote {output_path.resolve()} with {count} " + f"{'entry' if count == 1 else 'entries'}.") + + +def _wizard(stdscr, args: argparse.Namespace, parser: argparse.ArgumentParser + ) -> Optional[dict]: + """Run every TUI screen; return the collected settings, or None to abort. + + The wizard is driven by ``tui.Wizard`` as a stack of screen closures: + each screen shows one interactive widget and returns the next screen + (a closure), ``Wizard.BACK`` (Esc/q pressed — pop to the previous + screen), or the final settings dict. Only screens that actually render + are pushed, so Esc always lands on the previous real screen. A step + whose value is already provided by a flag (``--host``, ``--port``, + ``--families``, ...) or does not apply (e.g. the port-sync prompt when + the port did not change) is folded into the ``_after_*`` guards and + never becomes a screen. Esc on the first screen aborts the whole + wizard. + """ + + s: dict = {} + + def ask_confirm(question: str, default: bool) -> bool: + result = tui.confirm(stdscr, question, default=default, + cancel_value=_GO_BACK) + if result is _GO_BACK: + raise _GoBack() + return result + + def resolve_checkout(audiocpp_dir: Path) -> None: + """Validate the audio.cpp checkout and populate the wizard state ``s``.""" + audiocpp_dir = Path(audiocpp_dir).resolve() + try: + catalog = load_model_catalog(audiocpp_dir) + except NotADirectoryError as exc: + raise _TuiError(str(exc)) + if not catalog: + raise _TuiError(f"No TTS model families found in " + f"{audiocpp_dir}/model_specs; check the " + "checkout is up to date") + catalog_by_family = {entry["family"]: entry for entry in catalog} + output_path = args.output if args.output is not None \ + else audiocpp_dir / "server.json" + # Modify flow: an existing server.json seeds the wizard's screens + # instead of being overwritten from scratch (an explicit --force + # still starts fresh). + existing_config = load_server_config(output_path) \ + if not args.force else None + if existing_config is not None: + existing_selected, existing_tasks = \ + server_config_selections(existing_config, catalog) + else: + existing_selected, existing_tasks = {}, {} + s.update({ + "audiocpp_dir": audiocpp_dir, + "catalog": catalog, + "catalog_by_family": catalog_by_family, + "output_path": output_path, + "existing_config": existing_config, + "existing_selected": existing_selected, + "existing_tasks": existing_tasks, + "existing_host": existing_config.get("host") + if existing_config else None, + "existing_port": existing_config.get("port") + if existing_config else None, + "existing_backend": existing_config.get("backend") + if existing_config else None, + "existing_voice_dir": existing_config.get("voice_dir") + if existing_config else None, + "detected_backend": detect_backend(audiocpp_dir), + }) + + def _families_from_flag() -> None: + requested = [f.strip() for f in args.families.split(",") if f.strip()] + unknown = [f for f in requested if f not in s["catalog_by_family"]] + if unknown: + raise _TuiError( + f"Unknown family in --families: {', '.join(unknown)}. " + f"Available: {', '.join(s['catalog_by_family'])}") + chosen: Dict[str, List[dict]] = {} + family_keys: List[str] = [] + for family in requested: + if family not in family_keys: + family_keys.append(family) + chosen[family] = [opt for opt in package_dir_options( + s["catalog_by_family"][family]) if opt["recommended"]] + s["chosen"] = chosen + s["family_keys"] = family_keys + + def _compute_entries() -> None: + # Design task menu. Esc raises _GoBack, which the caller turns into + # Wizard.BACK (the design prompts are grouped: Esc returns to the + # families tree). + def task_picker(install_id: str) -> str: + result = tui.menu( + stdscr, + f"How should the '{install_id}' package be hosted?", + [ + ("design (vdes) - describe the voice with " + "--instructions", TASK_VDES), + ("tts - normal synthesis", TASK_TTS), + ], default_index=0, back_value=_GO_BACK) + if result is _GO_BACK: + raise _GoBack() + return result + + model_entries, entry_ids, install_guidance, \ + design_entry_ids, include_clone = _build_entries( + s["family_keys"], s["chosen"], s["catalog_by_family"], + task_picker, known_tasks=s["existing_tasks"]) + s.update({ + "model_entries": model_entries, + "entry_ids": entry_ids, + "install_guidance": install_guidance, + "design_entry_ids": design_entry_ids, + "include_clone": include_clone, + }) + + def _finalize() -> dict: + return { + "audiocpp_dir": s["audiocpp_dir"], + "catalog": s["catalog"], + "catalog_by_family": s["catalog_by_family"], + "output_path": s["output_path"], + "family_keys": s["family_keys"], + "chosen": s["chosen"], + "model_entries": s["model_entries"], + "entry_ids": s["entry_ids"], + "install_guidance": s["install_guidance"], + "design_entry_ids": s["design_entry_ids"], + "include_clone": s["include_clone"], + "host": s["host"], + "port": s["port"], + "backend": s["backend"], + "build": s["build"], + "lazy_load": s["lazy_load"], + "sync_port": s["sync_port"], + "sync_model_ids": s["sync_model_ids"], + "wav_dir": s["wav_dir"], + "plan": s["plan"], + "download": s["download"], + "delete_unused": s["delete_unused"], + "unused_entries": s["unused_entries"], + } + + def screen_families(): + """Pick TTS model families and packages (the modify tree).""" + tree_families = _models._build_tree_families(s["catalog"]) + # Modify flow: pre-check the models an existing server.json hosts, + # so the tree opens as a "modify" list rather than a fresh one. + checked_set = set() + for family, dirs in s["existing_selected"].items(): + if family not in s["catalog_by_family"]: + continue + family_index = s["catalog"].index(s["catalog_by_family"][family]) + valid_dirs = {opt["target_directory"] + for opt in package_dir_options( + s["catalog_by_family"][family])} + for target in dirs: + if target in valid_dirs: + checked_set.add((family_index, target)) + picked = tui.checkbox_tree( + stdscr, "Select TTS model families to host", + tree_families, expand_all=args.all_packages, + back_value=_GO_BACK, checked=checked_set) + if picked is _GO_BACK: + return tui.Wizard.BACK + chosen: Dict[str, List[dict]] = {} + family_keys: List[str] = [] + for family_index, option_key in picked: + family = s["catalog"][family_index]["family"] + if family not in chosen: + chosen[family] = [] + family_keys.append(family) + chosen[family].append(option_key) + for family in list(chosen): + keyed = {opt["target_directory"]: opt + for opt in package_dir_options( + s["catalog_by_family"][family])} + chosen[family] = [keyed[key] for key in chosen[family]] + s["chosen"] = chosen + s["family_keys"] = family_keys + return screen_host + + def _after_families(): + if args.families is not None: + _families_from_flag() + return screen_host + return screen_families + + def screen_host(): + """Build the model entries, then ask the bind host. + + The task/id pickers (when any) run here too and are grouped with + this screen: Esc on one of them (or on the host field) returns to + the families tree. + """ + try: + _compute_entries() + except _GoBack: + return tui.Wizard.BACK + if args.host is not None: + s["host"] = args.host + return _after_host() + host = tui.line_edit( + stdscr, "Bind host", + s["existing_host"] if isinstance(s["existing_host"], str) + else DEFAULT_HOST, + help_lines=["The IP address audiocpp will be hosted on", + "127.0.0.1 (this machine) is probably " + "correct"], back_value=_GO_BACK) + if host is _GO_BACK: + return tui.Wizard.BACK + s["host"] = host + return _after_host() + + def _after_host(): + if args.port is None: + return screen_port + s["port"] = args.port + return _after_port() + + def screen_port(): + port_text = tui.line_edit( + stdscr, "Port", + str(s["existing_port"]) if isinstance(s["existing_port"], int) + else str(_configsync.config_port()), + validate=lambda s: None if (s.isdigit() + and 1 <= int(s) <= 65535) + else "Enter a port number between 1 and 65535", + help_lines=["The port audiocpp will be hosted on"], + back_value=_GO_BACK) + if port_text is _GO_BACK: + return tui.Wizard.BACK + s["port"] = int(port_text) + return _after_port() + + def _after_port(): + s["sync_port"] = None + if s["port"] != _configsync.config_port(): + return screen_sync_port + return _after_sync() + + def screen_sync_port(): + sync_port = tui.confirm( + stdscr, "Update AUDIOCPP_API_URL in app/converter/config.py " + f"to port {s['port']} so audiobook.py talks to this server", + default=True, cancel_value=_GO_BACK) + if sync_port is _GO_BACK: + return tui.Wizard.BACK + s["sync_port"] = sync_port + return _after_sync() + + def _after_sync(): + if args.build_backend: + s["backend"] = args.build_backend + s["build"] = s["detected_backend"] is None + return _after_backend() + if args.backend: + s["backend"] = args.backend + s["build"] = False + return _after_backend() + if s["detected_backend"] is not None: + # Already built: use the detected backend, no menu, no build. + s["backend"] = s["detected_backend"] + s["build"] = False + return _after_backend() + # Not built for any backend yet: always ask which backend the server + # should use and offer to build it — even on a modify run, so a user + # who declined the build the first time is never stranded without a + # way to build from the TUI. + return screen_backend + + def screen_backend(): + # Pre-select the backend an existing server.json records (modify + # flow), so re-running setup lands on the previous choice. + backend_options, backend_default = _backend_options(None) + if s["existing_backend"] in BACKENDS: + backend_default = next( + (index for index, (_label, value) in enumerate(backend_options) + if value == s["existing_backend"]), backend_default) + backend = tui.menu( + stdscr, "Which inference backend should audiocpp_server " + "use?", backend_options, + default_index=backend_default, back_value=_GO_BACK) + if backend is _GO_BACK: + return tui.Wizard.BACK + s["backend"] = backend + if _build.built_server_binary(s["audiocpp_dir"], backend) is not None: + # A checkout with builds for several backends: this one is + # already built, so there is nothing to build. + s["build"] = False + return _after_backend() + return screen_build + + def screen_build(): + # Not built for the chosen backend yet: offer to build it now. The + # build itself runs in the TUI task view (or the console tail for + # CLI runs) after the wizard. + build = tui.confirm( + stdscr, f"audiocpp_server is not built for {s['backend']}. " + f"Build it now (runs scripts/build_*)?", + default=True, cancel_value=_GO_BACK) + if build is _GO_BACK: + return tui.Wizard.BACK + s["build"] = build + return _after_backend() + + def _after_backend(): + s["lazy_load"] = True + return _after_lazy() + + def _after_lazy(): + if args.input_dir is not None: + s["wav_dir"] = args.input_dir + return _after_wav() + if s["include_clone"]: + return screen_wav + s["wav_dir"] = None + return _after_wav() + + def screen_wav(): + wav_start = detect_wav_dir(s["audiocpp_dir"], TTS_ROOT) + # Modify flow: an existing voice_dir seeds the browser so the user + # can accept it on Enter instead of re-navigating. + if isinstance(s["existing_voice_dir"], str) and s["existing_voice_dir"]: + wav_start = Path(s["existing_voice_dir"]) + wav_dir = tui.browse_directory( + stdscr, "Select the directory with your .wav voices", + info=_wav_dir_info, preview=_wav_dir_preview, + start=wav_start if wav_start is not None else VOICES_DIR, + back_value=_GO_BACK) + if wav_dir is _GO_BACK: + return tui.Wizard.BACK + s["wav_dir"] = wav_dir + return _after_wav() + + def _after_wav(): + s["plan"] = None + if s["include_clone"] and s["wav_dir"] is not None: + wav_files = find_wav_files(s["wav_dir"]) + if wav_files: + prompt_path = s["wav_dir"] / PROMPT_TEXT_FILENAME + if prompt_path.exists() and not args.force: + return screen_transcription + existing = read_prompt_text(prompt_path) if ( + prompt_path.exists() and not args.force) else {} + s["plan"] = _voices._decide_transcription( + wav_files, existing, prompt_path.exists(), + args.force, ask_confirm) + return _after_transcription() + + def screen_transcription(): + # Transcription plan (questions only; transcription runs after). + wav_files = find_wav_files(s["wav_dir"]) + prompt_path = s["wav_dir"] / PROMPT_TEXT_FILENAME + existing = read_prompt_text(prompt_path) if ( + prompt_path.exists() and not args.force) else {} + try: + s["plan"] = _voices._decide_transcription( + wav_files, existing, prompt_path.exists(), + args.force, ask_confirm) + except _GoBack: + return tui.Wizard.BACK + return _after_transcription() + + def _after_transcription(): + s["sync_model_ids"] = None + if len(s["entry_ids"]) == 1 and not ( + config.AUDIOCPP_MODEL_ID == s["entry_ids"][0] + and config.AUDIOCPP_CLONE_MODEL_ID == s["entry_ids"][0]): + return screen_model_sync + return _after_model_sync() + + def screen_model_sync(): + sync_model_ids = tui.confirm( + stdscr, "Update AUDIOCPP_MODEL_ID and " + "AUDIOCPP_CLONE_MODEL_ID in app/converter/config.py to " + f"'{s['entry_ids'][0]}' so audiobook.py uses this model", + default=True, cancel_value=_GO_BACK) + if sync_model_ids is _GO_BACK: + return tui.Wizard.BACK + s["sync_model_ids"] = sync_model_ids + return _after_model_sync() + + def _after_model_sync(): + new_paths = {entry["path"] for entry in s["model_entries"]} + s["unused_entries"] = _models.unused_installed_entries( + s["output_path"], new_paths) \ + if s["existing_config"] is not None else [] + s["delete_unused"] = False + if s["unused_entries"]: + return screen_delete_unused + return _after_delete() + + def screen_delete_unused(): + delete_unused = tui.confirm( + stdscr, "Delete unused models?", default=False, + cancel_value=_GO_BACK) + if delete_unused is _GO_BACK: + return tui.Wizard.BACK + s["delete_unused"] = delete_unused + return _after_delete() + + def _after_delete(): + manager = s["audiocpp_dir"] / "tools" / "model_manager_v2.py" + if manager.is_file(): + return screen_download + s["download"] = False + return _finalize() + + def screen_download(): + # Automatic model download (or print the install commands). + try: + s["download"] = _models._decide_download( + s["audiocpp_dir"], s["model_entries"], ask_confirm) + except _GoBack: + return tui.Wizard.BACK + return _finalize() + + # First screen: resolve the checkout directly when it already exists + # (the modify flow), so the wizard starts on a real screen. When no + # checkout exists, clone it into ./app/audio.cpp (streaming inside the + # TUI task view, not by dropping to the console) without asking, then + # continue the same way. + audiocpp_dir = _build.find_local_checkout() + if audiocpp_dir is None: + target = APP_DIR / AUDIOCPP_DIR_NAME + rc = taskview.run_steps(stdscr, "Clone audio.cpp", [ + taskview.TaskStep( + f"Cloning audio.cpp into {target}", + lambda emit, cancel: common.git_clone( + AUDIOCPP_GIT_URL, target, emit=emit, cancel=cancel)), + taskview.TaskStep( + "Apply ggml build patches", + lambda emit, cancel: _build.apply_ggml_patches( + target, emit=emit, cancel=cancel)), + ]) + if rc == 130: + # Cancelled from the task view: abort the wizard quietly. + return None + if rc != 0: + raise _TuiError( + f"audio.cpp setup step failed (exit {rc}). Clone " + f"audio.cpp manually: git clone " + f"{AUDIOCPP_GIT_URL} {target}, then re-run") + audiocpp_dir = target + resolve_checkout(audiocpp_dir) + first = _after_families() + return tui.Wizard().run(first) + + +def _execute_lanes(settings: dict, + args: argparse.Namespace) -> List[taskview.TaskLane]: + """Build the ordered setup steps for the in-TUI task view, per lane. + + The same work ``_execute`` runs on the console, split into two lanes so + the view can run the build in one pane while configuring and downloading + models in the other (both progress bars visible at once). The build lane + exists only when ``settings["build"]`` is set; the models lane always + exists (transcribe → write server.json → download/print commands). + Shared results (the transcription mapping) travel through a small closure + dict scoped to the models lane. Each step's ``work(emit, cancel)`` + returns its exit code; subprocess steps stream through EMIT and abort on + CANCEL, while print()-based steps are captured by the view's stdout + routing. + """ + audiocpp_dir = settings["audiocpp_dir"] + state: dict = {} + build = settings.get("build") + lanes: List[taskview.TaskLane] = [] + + if build: + def build_step(emit, cancel): + rc = _build.build_audiocpp(audiocpp_dir, settings["backend"], + emit=emit, cancel=cancel) + if rc != 0: + print(f"[WARNING] build exited with code {rc}; the server.json " + "was still written — build audiocpp_server manually " + "before starting it") + else: + print("[OK] build complete") + return rc + lanes.append(taskview.TaskLane( + "Build", + [taskview.TaskStep( + f"Build audiocpp_server ({settings['backend']})", + build_step)])) + + def transcribe(emit, cancel): + args.input_dir = settings["wav_dir"] + if settings["include_clone"] and args.input_dir is not None: + transcripts, write_prompt = _voices._transcribe( + args, plan=settings["plan"], cancel=cancel) + elif args.input_dir is not None: + print(f"[WARNING] Ignoring {args.input_dir}: no clone-capable " + "family selected, so voice presets are not used") + transcripts, write_prompt = {}, False + else: + transcripts, write_prompt = {}, False + state["transcripts"] = transcripts + state["write_prompt"] = write_prompt + return 0 + + def write(emit, cancel): + # Port sync (applied now that the terminal is back). + if settings["sync_port"] is True: + _configsync._apply_port_sync(settings["port"], True) + elif settings["sync_port"] is False: + _configsync._apply_port_sync(settings["port"], False) + + _write_and_advise( + audiocpp_dir, settings["wav_dir"], settings["output_path"], + settings["model_entries"], settings["install_guidance"], + settings["host"], settings["port"], settings["backend"], + settings["lazy_load"], state["transcripts"], state["write_prompt"]) + + # Delete-unused cleanup (modify flow): remove the already-downloaded + # models the new selection dropped. The regenerated server.json + # already only lists the kept entries. + if settings.get("delete_unused"): + removed = _models.delete_model_files(settings["output_path"], + settings["unused_entries"]) + print(f"[OK] Deleted {removed} unused model " + f"{'entry' if removed == 1 else 'entries'} from disk.") + + if len(settings["entry_ids"]) == 1: + _configsync._offer_config_model_id_sync(settings["entry_ids"][0], + settings["sync_model_ids"]) + _voices.print_empty_transcript_warning(state["transcripts"]) + return 0 + + def install(emit, cancel): + _models._install_models(audiocpp_dir, settings["install_guidance"], + settings["download"], emit=emit, cancel=cancel) + _build._print_launch_hint(audiocpp_dir, settings["output_path"]) + return 0 + install_title = "Download models" if settings.get("download") \ + else "Print model install commands" + + lanes.append(taskview.TaskLane( + "Configure & download", + [taskview.TaskStep("Transcribe reference voices", transcribe), + taskview.TaskStep("Write server.json & sync config", write), + taskview.TaskStep(install_title, install)])) + + return lanes + + +def _execute_steps(settings: dict, + args: argparse.Namespace) -> List[taskview.TaskStep]: + """The ordered setup steps for the sequential console path. + + The lanes ``_execute_lanes`` builds, flattened into one ordered list + (build first, then transcribe → write → download), so the console tail + is byte-identical to the pre-lanes behavior. + """ + steps: List[taskview.TaskStep] = [] + for lane in _execute_lanes(settings, args): + steps.extend(lane.steps) + return steps + + +def _execute(settings: dict, args: argparse.Namespace) -> int: + """Shared console tail: build, sync, transcribe, write, install, advise. + + Runs after the TUI wizard returns (or after _collect_from_flags for a + non-interactive run): the terminal is plain, so subprocess output and + transcription progress appear normally. The same work as + ``_execute_steps``, run with no emit (console streaming). + """ + return taskview.run_steps_inline(_execute_steps(settings, args)) + + +def setup_screen(stdscr) -> int: + """Run the setup wizard on an existing curses screen (the hub's). + + The hub drives this as one screen of its own ``tui.Wizard`` stack, so + Esc on the wizard's first screen simply returns here and the hub pops + back to the menu that launched it. The setup tail (build, transcribe, + write, download) runs inside the TUI task view on this same screen, so + the hub's curses session stays intact and the user sees per-step status + and progress instead of being dropped to the console. On a fresh install + the build and the model setup run as two parallel lanes (a split view), + so cloning → configuring → building+downloading is one continuous, + one-click flow; the individual "Build" and "Download Missing Models" hub + actions remain only as fallbacks when something fails or is interrupted. + Returns 0 on completion, 1 when the user aborted. + """ + parser = build_parser() + args = parser.parse_args([]) + settings = _wizard(stdscr, args, parser) + if settings is None: + return 1 + return taskview.run_lanes(stdscr, "Setting up audio.cpp", + _execute_lanes(settings, args)) + + +def build_screen(stdscr) -> int: + """Build audiocpp_server from the hub when the checkout has no binary. + + Asks which backend to build for (pre-selecting the backend an existing + server.json records, else cuda), runs the build inside the TUI task view + — alongside a download of any missing models when server.json is already + configured and those models map to an install command (the split view), + or just the build otherwise — then updates server.json's ``backend`` + field to match. Returns 0 on success, non-zero when the user backed out, + cancelled, or the build failed. This is the hub's "Build audio.cpp + server" action, so a checkout that was cloned but never built is always + buildable from the TUI; the standalone "Download Missing Models" action + stays as the fallback when the download fails or is interrupted. + """ + checkout = _build.find_local_checkout() + if checkout is None: + tui.flash(stdscr, "No audio.cpp checkout found — install audio.cpp " + "first.", "err") + return 1 + if _build.find_audiocpp_server_bin(checkout) is not None: + tui.flash(stdscr, "audiocpp_server is already built.", "ok") + return 0 + server_config = load_server_config(checkout / "server.json") or {} + recorded = server_config.get("backend") + options, default = _backend_options(None) + if recorded in BACKENDS: + default = next((i for i, (_label, value) in enumerate(options) + if value == recorded), default) + backend = tui.menu( + stdscr, "Which inference backend should audiocpp_server be built " + "for?", options, default_index=default, back_value=_GO_BACK) + if backend is _GO_BACK: + return 1 + + def build_step(emit, cancel): + return _build.build_audiocpp(checkout, backend, emit=emit, cancel=cancel) + + lanes = [taskview.TaskLane( + "Build", [taskview.TaskStep( + f"Build audiocpp_server ({backend})", build_step)])] + + # Missing models this build can also fetch, so a configured backend that + # lost its binary is restored to "installed" in one step. + server_json = checkout / "server.json" + missing = _models.missing_model_entries(server_json) if server_json.exists() else [] + guidance = _models.missing_model_install_guidance(checkout, missing) \ + if missing else [] + + if guidance: + def download_step(emit, cancel): + _models.install_models(checkout, guidance, emit=emit, cancel=cancel) + return 0 + lanes.append(taskview.TaskLane( + "Download models", + [taskview.TaskStep("Download missing models", download_step)])) + + title = "Build & download models" if len(lanes) == 2 \ + else "Build audiocpp_server" + rc = taskview.run_lanes(stdscr, title, lanes) + if rc != 0: + return rc + if _configsync.update_server_backend(backend): + tui.flash(stdscr, f"audiocpp_server built for {backend}.", "ok") + else: + tui.flash(stdscr, f"audiocpp_server built for {backend}. (Could not " + "update server.json's backend field — reconfigure audio.cpp " + "if it was already configured.)", "warn") + # Models that can't be mapped to an install command still need hand + # installation; say so now rather than leaving the user in the dark. + if missing and not guidance: + tui.flash(stdscr, _models.hand_install_guidance(checkout, missing), "err") + return 0 + + +def run_tui(args: Optional[argparse.Namespace] = None, + parser: Optional[argparse.ArgumentParser] = None) -> int: + """Run the audio.cpp setup wizard end-to-end. + + With no ARGS (the hub's call) a default namespace is built so the full + wizard runs. Called from ``main`` after argparse when the terminal is + interactive. Returns the process exit code. + """ + import curses + if args is None: + parser = build_parser() + args = parser.parse_args([]) + if args.input_dir is not None and not args.input_dir.is_dir(): + print(f"[ERROR] --wavs not found: {args.input_dir}", + file=sys.stderr) + return 2 + try: + settings = curses.wrapper(_wizard, args, parser) + except _TuiError as exc: + print(f"[ERROR] {exc}", file=sys.stderr) + return 2 + except tui.WizardCancelled: + print("\n[INFO] Cancelled; nothing was written") + return 1 + try: + curses.curs_set(1) # restore the text cursor hidden by the TUI + except curses.error: + pass + if settings is None: + print("[INFO] Aborted; existing server.json kept") + return 1 + return _execute(settings, args) + + +def _collect_from_flags(args: argparse.Namespace, + parser: argparse.ArgumentParser) -> Optional[dict]: + """Build the settings dict from flags for a non-interactive run. + + Every required value must come from a flag (there are no prompts in a + non-interactive run); a missing one is a hard ``parser.error``. Returns + the settings dict, or None when the user declined an overwrite (the + default-location fallback then also exists). + """ + # Checkout: ./app/audio.cpp, else --clone clones one there. + audiocpp_dir = _build.find_local_checkout() + if audiocpp_dir is None and args.clone: + target = APP_DIR / AUDIOCPP_DIR_NAME + rc = common.git_clone(AUDIOCPP_GIT_URL, target) + if rc != 0: + parser.error(f"git clone failed (exit {rc}); clone audio.cpp " + f"manually: git clone {AUDIOCPP_GIT_URL} {target}") + patch_rc = _build.apply_ggml_patches(target) + if patch_rc != 0: + parser.error( + f"ggml build patches could not be applied to {target} " + f"(exit {patch_rc}); see messages above. The audio.cpp " + f"fork's vendored ggml may have changed — re-evaluate " + f"app/backends/patches/.") + audiocpp_dir = target + if audiocpp_dir is None: + parser.error( + "An audio.cpp checkout is required. Pass --clone to clone " + "app/audio.cpp, or run without flags for the TUI wizard.") + try: + catalog = load_model_catalog(audiocpp_dir) + except NotADirectoryError as exc: + parser.error(str(exc)) + if not catalog: + parser.error( + f"No TTS model families found in {audiocpp_dir}/model_specs; " + "check the checkout is up to date") + catalog_by_family = {entry["family"]: entry for entry in catalog} + + # Families: required from --families in a non-interactive run. + if args.families is None: + parser.error("--families is required in a non-interactive run (or run " + "without flags for the TUI wizard)") + requested = [f.strip() for f in args.families.split(",") if f.strip()] + unknown = [f for f in requested if f not in catalog_by_family] + if unknown: + parser.error( + f"Unknown family in --families: {', '.join(unknown)}. " + f"Available: {', '.join(catalog_by_family)}") + family_keys: List[str] = [] + for fam in requested: + if fam not in family_keys: + family_keys.append(fam) + + chosen: Dict[str, List[dict]] = {} + for family in family_keys: + opts = package_dir_options(catalog_by_family[family]) + if args.all_packages: + chosen[family] = opts + else: + chosen[family] = [opt for opt in opts if opt["recommended"]] + + # Non-interactive picker: design packages default to vdes. + def task_picker(install_id: str) -> str: + return TASK_VDES + + model_entries, entry_ids, install_guidance, design_entry_ids, include_clone = \ + _build_entries(family_keys, chosen, catalog_by_family, + task_picker) + + # Server settings. + host = args.host or DEFAULT_HOST + detected_backend = detect_backend(audiocpp_dir) + if args.build_backend: + backend = args.build_backend + build = detected_backend is None + elif args.backend: + backend = args.backend + build = False + elif detected_backend is not None: + backend = detected_backend + build = False + else: + backend = "cuda" + build = False + port = args.port if args.port is not None else _configsync.config_port() + lazy_load = True + + # Output path / overwrite (decline falls back to cwd, then aborts). + output_path = args.output if args.output is not None \ + else audiocpp_dir / "server.json" + if output_path.exists() and not args.force: + if args.output is None: + output_path = Path.cwd() / "server.json" + if output_path.exists() and not args.force: + print("[INFO] Aborted; existing server.json kept") + return None + else: + print("[INFO] Aborted; existing server.json kept") + return None + + # Config sync decisions (auto-apply unless explicitly declined). + sync_port: Optional[bool] = None + if port != _configsync.config_port(): + sync_port = not args.no_sync_port + sync_model_ids: Optional[bool] = None + if len(entry_ids) == 1 and not ( + config.AUDIOCPP_MODEL_ID == entry_ids[0] + and config.AUDIOCPP_CLONE_MODEL_ID == entry_ids[0]): + sync_model_ids = not args.no_sync_model_ids + + # Wav dir + transcription plan (defaults to the project's voices/ dir). + wav_dir = args.input_dir if args.input_dir is not None else VOICES_DIR + plan: Optional[dict] = None + if include_clone and wav_dir is not None: + wav_files = find_wav_files(wav_dir) + if wav_files: + prompt_path = wav_dir / PROMPT_TEXT_FILENAME + plan = _voices._flag_plan(wav_files, prompt_path, args.force) + + return { + "audiocpp_dir": audiocpp_dir, + "catalog": catalog, + "catalog_by_family": catalog_by_family, + "output_path": output_path, + "family_keys": family_keys, + "chosen": chosen, + "model_entries": model_entries, + "entry_ids": entry_ids, + "install_guidance": install_guidance, + "design_entry_ids": design_entry_ids, + "include_clone": include_clone, + "host": host, + "port": port, + "backend": backend, + "build": build, + "lazy_load": lazy_load, + "sync_port": sync_port, + "sync_model_ids": sync_model_ids, + "wav_dir": wav_dir, + "plan": plan, + "download": args.download, + } + + +def build_parser() -> argparse.ArgumentParser: + """The audio.cpp setup CLI (also used to build a default namespace).""" + parser = argparse.ArgumentParser( + description="Set up the audio.cpp TTS backend: clone/build, pick " + "models, write server.json, and sync app/converter/config.py.") + parser.add_argument("--wavs", type=resolve_wav_dir_arg, default=None, + dest="input_dir", metavar="WAV_DIR", + help="Directory with .wav reference files to publish as " + "a server-level voice_dir cloning library " + f"(default: {VOICES_DIR}; asked for when omitted " + "in the TUI)") + parser.add_argument("--output", type=Path, default=None, + help="Output path for server.json (default: " + "server.json inside the audio.cpp checkout; an " + "existing file is overwritten only with --force " + "or a TUI confirm)") + parser.add_argument("--clone", action="store_true", + help="Non-interactive: clone audio.cpp into " + "./app/audio.cpp when no checkout is found") + parser.add_argument("--families", type=str, default=None, + help="Comma-separated model families to host, as named " + "in the audio.cpp catalog (e.g. " + "qwen3_tts,higgs_audio_tts). Required in a " + "non-interactive run; skips the family tree in " + "the TUI") + parser.add_argument("--all-packages", action="store_true", + help="Host every installable package of each selected " + "family (distinct target_directory) instead of " + "only the recommended one. Voice-design packages " + "are hosted with task 'vdes'") + parser.add_argument("--host", type=str, default=None, + help="Bind host for the server (default: 127.0.0.1)") + parser.add_argument("--port", type=int, default=None, + help="Port for the server (default: the port in " + "AUDIOCPP_API_URL from app/converter/config.py)") + parser.add_argument("--backend", choices=BACKENDS, default=None, + help="Inference backend recorded in server.json " + "(default: auto-detected from the checkout's " + "build/ directory, else cuda)") + parser.add_argument("--build-backend", choices=BACKENDS, default=None, + help="Build audiocpp_server for this backend when it " + "is not built yet, and use it in server.json") + parser.add_argument("--whisper-model", type=str, default="base", + help="Whisper model size for transcription " + "(default: base)") + parser.add_argument("--force", action="store_true", + help="Overwrite the output file (and prompt_text) " + "without prompting; in the TUI, start the " + "wizard fresh instead of loading the existing " + "server.json") + parser.add_argument("--download", action="store_true", + help="Run model_manager_v2.py install for each hosted " + "model automatically (default: print the commands " + "only)") + parser.add_argument("--no-sync-port", action="store_true", + help="Do not rewrite AUDIOCPP_API_URL in " + "app/converter/config.py when --port differs") + parser.add_argument("--no-sync-model-ids", action="store_true", + help="Do not rewrite AUDIOCPP_MODEL_ID/" + "AUDIOCPP_CLONE_MODEL_ID for a single-entry server") + return parser + + +def main() -> int: + parser = build_parser() + args = parser.parse_args() + + if args.input_dir is not None and not args.input_dir.is_dir(): + parser.error( + f"WAV directory not found: {args.input_dir}\n" + f" (resolved from the current working directory: " + f"{Path.cwd()})\n" + " --wavs must be a directory containing the .wav " + "reference files to use as voice cloning presets") + + if _interactive(): + return run_tui(args, parser) + + # Non-interactive (no terminal, or all flags supplied): flag-only path. + settings = _collect_from_flags(args, parser) + if settings is None: + return 1 + return _execute(settings, args) + + |
