aboutsummaryrefslogtreecommitdiff
path: root/tools/make_audiocpp_server_json.py
diff options
context:
space:
mode:
Diffstat (limited to 'tools/make_audiocpp_server_json.py')
-rwxr-xr-xtools/make_audiocpp_server_json.py1841
1 files changed, 0 insertions, 1841 deletions
diff --git a/tools/make_audiocpp_server_json.py b/tools/make_audiocpp_server_json.py
deleted file mode 100755
index 3446428..0000000
--- a/tools/make_audiocpp_server_json.py
+++ /dev/null
@@ -1,1841 +0,0 @@
-#!/usr/bin/env python3
-"""Interactively generate a server.json for the audio.cpp audiocpp_server.
-
-Reads the model catalog (``model_specs/*.json``) from a local audio.cpp
-checkout and offers every TTS model family audio.cpp supports, so one
-server.json can host several lazily-loaded model entries at once. The
-converter itself is family-agnostic (it detects the family of the selected
-entry from ``GET /v1/models`` at startup), so any TTS family listed in the
-catalog works without further changes.
-
-By default the tool runs as a colorful full-screen TUI (curses): every
-screen is a centered DOS-style dialog on a black desktop — a file
-browser for the audio.cpp checkout and the .wav directory, an
-expandable checkbox tree of model families and their installable
-packages, centered single-question screens for the server settings,
-and Yes/No buttons for every confirmation. In the checkout browser,
-pressing Enter (or Right) on a subdirectory named ``audio.cpp`` that
-already contains ``model_specs/`` picks it directly, skipping the
-``[ Use this directory ]`` step; pressing Esc on the overwrite
-confirmation then returns to the browser inside that checkout (with
-the auto-pick disabled), instead of aborting the wizard. Esc on any
-other wizard screen falls back to the previous screen group (only the
-first screen, the checkout browser, exits on Esc). Pass ``--notui`` to
-use the classic numbered line prompts instead (also selected
-automatically when stdin/stdout is not a terminal, or when curses is
-unavailable such as on Windows without ``windows-curses``). Every
-value can also be supplied as a command-line flag, which skips the
-corresponding screen or prompt.
-
-Each family is hosted through its recommended package by default; the TUI
-tree always lists every installable package (distinct ``target_directory``
-values) as checkboxes, while ``--all-packages`` in prompt mode offers a
-per-family package checklist (and pre-expands every family in the TUI).
-Packages whose name marks them as voice-design models are asked whether to
-host them with task "vdes" (describe the voice with ``--instructions``) or
-plain "tts". All families are treated equally and listed alphabetically.
-
-The .wav directory browser (and the prompt default) starts in the single
-directory that directly contains .wav files across the audio.cpp checkout
-and the tts-audiobook-generator root, if exactly one exists; the
-generator's ``output/`` directory is never offered.
-
-Cloning reference .wav files (``--wavs DIR``) are transcribed with a local
-Whisper backend (faster_whisper or whisper) and published as a server-level
-``voice_dir`` plus a ``prompt_text`` mapping file written into the wav
-directory, so every hosted clone-capable family can use them with
-``--voice``. If ``prompt_text`` already exists, only voices that are missing
-(or have an empty transcript) are re-transcribed, and you are asked first
-when everything is already transcribed or when a mix of existing and new
-voices is detected. Transcription runs in the plain console after the TUI
-has gathered every setting.
-
-Usage:
- python tools/make_audiocpp_server_json.py [--wavs WAV_DIR]
- [--output PATH] [--audiocpp-dir PATH] [--families FAM1,FAM2]
- [--all-packages] [--host HOST] [--port PORT]
- [--backend {cuda,vulkan,hip,cpu}] [--lazy-load]
- [--whisper-model NAME] [--force] [--notui]
-
---wavs is the directory of .wav reference files used as voice cloning
-presets; when omitted it is asked for. It is checked up front and reported
-with its resolved absolute path if it does not exist.
-
-server.json is written into the audio.cpp checkout by default (next to
-model_specs/). If that file already exists you are prompted [Y/n] before
-overwriting; answering "n" writes server.json in the current working
-directory instead (in the TUI, Esc on that prompt returns to the
-checkout browser rather than aborting). After a successful run the
-console output is the written file plus one copy-pasteable
-model_manager_v2.py install command per hosted model; you are also
-asked whether to run those downloads automatically.
-
---audiocpp-dir defaults to a detected audio.cpp checkout (the AUDIOCPP_DIR
-environment variable, or an ``audio.cpp`` directory next to or above the
-current working directory); if none is found it is asked interactively. The
-checkout must contain a ``model_specs/`` directory. A leading ``~`` in a
-path argument or prompt answer is expanded.
-
---backend is the inference backend audiocpp_server was built for. When the
-checkout contains a build directory (``build/<platform>-<backend>-<type>``
-with a built ``bin/audiocpp_server``), that backend is auto-detected,
-selected by default and marked ``[auto-detected]`` in the menu.
-"""
-
-import argparse
-import json
-import os
-import re
-import subprocess
-import sys
-import urllib.parse
-from pathlib import Path
-from typing import Callable, Dict, List, Optional, Set, Tuple
-
-# Allow running from any working directory.
-sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
-
-from converter import config
-from converter.tts import transcribe_reference_audio, whisper_backend_available
-
-DEFAULT_HOST = "127.0.0.1"
-FALLBACK_PORT = 8080
-CONFIG_PATH = Path(__file__).resolve().parent.parent / "converter" / "config.py"
-
-# The tts-audiobook-generator checkout root (where audiobook.py lives), used
-# to default the .wav directory browser. The audio.cpp checkout is detected
-# separately (see detect_audiocpp_dir).
-TTS_ROOT = Path(__file__).resolve().parent.parent
-# Output directory of tts-audiobook-generator; never offered as a .wav source.
-TTS_OUTPUT_DIR = "output"
-
-BACKENDS = ("cuda", "vulkan", "hip", "cpu")
-
-PROMPT_TEXT_FILENAME = "prompt_text"
-TASK_TTS = "tts"
-TASK_VDES = "vdes"
-
-# Sentinel returned by tui.confirm (via its cancel_value) when the user
-# presses Esc on an overwrite prompt to go back to the checkout browser
-# instead of aborting the wizard.
-_GO_BACK = object()
-
-
-class _GoBack(Exception):
- """Raised inside the TUI wizard to fall back to the previous screen group.
-
- Every wizard widget is passed ``back_value=_GO_BACK`` so Esc returns the
- sentinel instead of aborting; pickers and confirmations that call into
- callbacks (task/id pickers, the transcription plan, the download prompt)
- convert that sentinel into this exception so the enclosing step can catch
- it and step back. Only the first screen (the checkout browser) lets Esc
- abort the whole wizard.
- """
-
-# Package names that mark a voice-design model (hosted with task "vdes").
-DESIGN_PACKAGE_RE = re.compile(r"voice[\s_\-]?design", re.IGNORECASE)
-
-# Short, friendly default entry ids for selected families. Other families
-# derive an id from their family name (see default_model_id). All families
-# are listed equally, in alphabetical order.
-PREFERRED_IDS = {
- "qwen3_tts": "qwen",
- "higgs_audio_tts": "higgs",
- "voxcpm2": "voxcpm2",
- "index_tts2": "indextts2",
-}
-
-
-class _TuiError(Exception):
- """A fatal error raised from inside the TUI wizard.
-
- The message is reported to stderr after the terminal is restored; the
- process exits with code 2 (matching a parser error).
- """
-
-
-def _curses_importable() -> bool:
- """Return True when the curses module can be imported."""
- try:
- import curses # noqa: F401
- return True
- except ImportError:
- return False
-
-
-def _load_tui():
- """Import the TUI widget module (tools/tui.py)."""
- try:
- from tools import tui
- except ImportError: # executed directly from the tools/ directory
- import tui
- return tui
-
-
-def _tui_enabled(args: argparse.Namespace) -> bool:
- """Decide whether to run the TUI or fall back to line prompts."""
- if args.notui:
- return False
- if not _curses_importable():
- return False
- try:
- return sys.stdin.isatty() and sys.stdout.isatty()
- except (AttributeError, ValueError):
- return False
-
-
-def normalize_dir_arg(value: str) -> Path:
- """Normalize a user-supplied path argument.
-
- Strips surrounding quotes (a common copy-paste artifact), expands a
- leading ``~``, and resolves the result to an absolute path so relative
- paths are always validated against the current working directory.
- """
- cleaned = value.strip()
- if len(cleaned) >= 2 and cleaned[0] == cleaned[-1] and cleaned[0] in "\"'":
- cleaned = cleaned[1:-1]
- return Path(os.path.expanduser(cleaned)).resolve()
-
-
-def resolve_wav_dir_arg(value: str) -> Path:
- """Normalize a user-supplied wav directory argument."""
- return normalize_dir_arg(value)
-
-
-def find_wav_files(input_dir: Path) -> list:
- """Return the .wav files in INPUT_DIR, sorted alphabetically by name."""
- return sorted(
- (path for path in input_dir.iterdir()
- if path.is_file() and path.suffix.lower() == ".wav"),
- key=lambda path: path.name.lower(),
- )
-
-
-def _count_wavs(directory: Path) -> int:
- """Count the .wav files in DIRECTORY (0 when it cannot be read)."""
- try:
- return sum(1 for path in directory.iterdir()
- if path.is_file() and path.suffix.lower() == ".wav")
- except OSError:
- return 0
-
-
-def detect_wav_dir(audiocpp_dir: Path, tts_root: Path) -> Optional[Path]:
- """Find a unique directory that directly contains .wav files.
-
- Looks shallowly (the root itself and its immediate subdirectories) in
- both the audio.cpp checkout and the tts-audiobook-generator root (where
- audiobook.py lives), since clone reference .wavs commonly live in either.
- The tts-audiobook-generator ``output/`` directory is excluded. When
- exactly one candidate is found it is returned (as a starting directory
- for the .wav browser); when none or several are found None is returned
- so the caller falls back to its default start location.
- """
- candidates: List[Path] = []
- seen: Set[Path] = set()
-
- def consider(directory: Path) -> None:
- try:
- resolved = directory.resolve()
- except OSError:
- return
- if resolved in seen:
- return
- seen.add(resolved)
- if _count_wavs(directory) > 0:
- candidates.append(directory)
-
- for root in (audiocpp_dir, tts_root):
- if not root.is_dir():
- continue
- consider(root)
- try:
- children = sorted(root.iterdir(), key=lambda p: p.name.lower())
- except OSError:
- continue
- for child in children:
- if not child.is_dir() or child.name.startswith("."):
- continue
- # Exclude the tts-audiobook-generator output directory.
- if root == tts_root and child.name == TTS_OUTPUT_DIR:
- continue
- consider(child)
-
- if len(candidates) == 1:
- return candidates[0]
- return None
-
-
-def _wav_dir_info(directory: Path) -> Tuple[str, str]:
- """TUI status describing the directory listed in the wav browser."""
- count = _count_wavs(directory)
- if count:
- wavs = ".wav" if count == 1 else ".wavs"
- return (f"{count} {wavs} found in this directory. Press Enter.",
- "ok")
- return ("No .wav files found in this directory", "warn")
-
-
-def _wav_dir_preview(directory: Path) -> Tuple[str, str]:
- """TUI status describing a highlighted subdirectory in the wav browser."""
- count = _count_wavs(directory)
- if count:
- wavs = ".wav" if count == 1 else ".wavs"
- return (f"{count} {wavs}", "ok")
- return ("no .wav files", "info")
-
-
-def _resolve_audiocpp_root(directory: Path) -> Optional[Path]:
- """Return the audio.cpp checkout root for DIRECTORY, or None.
-
- Accepts either the checkout root itself (it must contain a
- ``model_specs`` directory) or the ``model_specs`` directory inside
- it (the parent is used), so the file browser cannot pick the wrong
- one of the two.
- """
- if (directory / "model_specs").is_dir():
- return directory
- if directory.name == "model_specs" and directory.is_dir():
- return directory.parent
- return None
-
-
-def _audiocpp_root_status(directory: Path) -> Tuple[str, str]:
- """TUI status describing the directory listed in the checkout browser."""
- if _resolve_audiocpp_root(directory) is not None:
- return ("model_specs/ found here", "ok")
- return ("No model_specs/ directory here", "warn")
-
-
-def _audiocpp_root_preview(directory: Path) -> Optional[Tuple[str, str]]:
- """TUI status for a highlighted subdirectory in the checkout browser."""
- if (directory / "model_specs").is_dir():
- return ("contains model_specs/", "ok")
- return None
-
-
-def _checkout_auto_select(entry: Path) -> Optional[Path]:
- """Auto-accept a highlighted checkout in the TUI browser.
-
- A subdirectory named ``audio.cpp`` that already contains a
- ``model_specs`` directory is the audio.cpp checkout root, so it is
- accepted immediately on Enter/Right (as if ``[ Use this directory ]``
- had been pressed) instead of being descended into. Anything else
- returns None so the user keeps browsing. This is only consulted
- while auto-accepting is still enabled; after the user presses Esc to
- go back, the browser is restarted inside the previously accepted
- checkout and this callback is no longer passed, so a wrong guess can
- be corrected.
- """
- if entry.name == "audio.cpp" and (entry / "model_specs").is_dir():
- return entry
- return None
-
-
-def ask(prompt: str, default: Optional[str] = None) -> Optional[str]:
- """Prompt for a free-text value with a default; EOF returns the default."""
- suffix = f" [{default}]" if default is not None else ""
- try:
- answer = input(f"{prompt}{suffix}: ").strip()
- except EOFError:
- return default
- return answer or default
-
-
-def ask_bool(prompt: str, default: bool = False) -> bool:
- """Prompt for a yes/no answer; Enter or EOF accepts the default."""
- suffix = " [Y/n]" if default else " [y/N]"
- while True:
- try:
- answer = input(f"{prompt}{suffix}: ").strip().lower()
- except EOFError:
- return default
- if not answer:
- return default
- if answer in ("y", "yes"):
- return True
- if answer in ("n", "no"):
- return False
- print("Please answer 'y' or 'n'.")
-
-
-def ask_port(default: int) -> int:
- """Prompt for a port number; Enter or EOF accepts the default."""
- while True:
- try:
- answer = input(f"Port [{default}]: ").strip()
- except EOFError:
- return default
- if not answer:
- return default
- try:
- value = int(answer)
- except ValueError:
- value = None
- if value is not None and 1 <= value <= 65535:
- return value
- print("Please enter a port number between 1 and 65535.")
-
-
-def ask_menu(title: str, options: list, default_index: int = 1) -> str:
- """Show a numbered menu and return the chosen option's value."""
- print(title)
- for number, (label, _) in enumerate(options, 1):
- print(f" {number}) {label}")
- while True:
- try:
- answer = input(f"Choice [{default_index}]: ").strip()
- except EOFError:
- return options[default_index - 1][1]
- if not answer:
- return options[default_index - 1][1]
- if answer.isdigit() and 1 <= int(answer) <= len(options):
- return options[int(answer) - 1][1]
- print(f"Please enter a number between 1 and {len(options)}.")
-
-
-def ask_checklist(title: str, options: list, default: Set[str]) -> Set[str]:
- """Show a numbered multi-select checklist and return the chosen values.
-
- Input is comma/space-separated numbers; Enter or EOF selects every option
- in DEFAULT. At least one option is required.
- """
- print(title)
- for number, (label, _) in enumerate(options, 1):
- print(f" {number}) {label}")
- default_numbers = [str(number) for number, (_, value) in enumerate(options, 1)
- if value in default]
- suffix = f" [{', '.join(default_numbers)}]"
- while True:
- try:
- answer = input(f"Choice{suffix}: ").strip()
- except EOFError:
- return set(default)
- if not answer:
- return set(default)
- parts = [p for p in re.split(r"[,\s]+", answer) if p]
- indices: List[int] = []
- valid = True
- for part in parts:
- if part.isdigit() and 1 <= int(part) <= len(options):
- indices.append(int(part))
- else:
- valid = False
- break
- if valid and indices:
- return {options[index - 1][1] for index in indices}
- print(f"Please enter comma-separated numbers between 1 and {len(options)}.")
-
-
-# Backend display order, with short descriptions. The backend name is padded
-# so the descriptions' dashes line up in the menu.
-_BACKEND_DESCRIPTIONS = (
- ("cuda", "NVIDIA GPUs (fastest)"),
- ("vulkan", "cross-vendor GPU"),
- ("hip", "AMD GPUs"),
- ("cpu", "no GPU required"),
-)
-
-
-def _backend_options(detected: Optional[str] = None
- ) -> Tuple[List[Tuple[str, str]], int]:
- """Build the aligned backend menu options and the default index.
-
- The backend names are padded to a common width so the ``-`` dashes
- before the descriptions line up. When DETECTED matches one of the
- options, that option gets ``[auto-detected]`` appended and is the
- default (cursor/start) selection; otherwise the first option is the
- default as before. Returns (options, default_index).
- """
- width = max(len(name) for name, _ in _BACKEND_DESCRIPTIONS)
- options: List[Tuple[str, str]] = []
- default_index = 0
- for index, (name, desc) in enumerate(_BACKEND_DESCRIPTIONS):
- label = f"{name.ljust(width)} - {desc}"
- if detected == name:
- label += " [auto-detected]"
- default_index = index
- options.append((label, name))
- return options, default_index
-
-
-def ask_backend(detected: Optional[str] = None) -> str:
- options, default_index = _backend_options(detected)
- return ask_menu(
- "Which inference backend was audiocpp_server built for?",
- options, default_index=default_index + 1)
-
-
-def config_port() -> int:
- """Return the port of AUDIOCPP_API_URL in converter/config.py."""
- try:
- return urllib.parse.urlsplit(config.AUDIOCPP_API_URL).port or FALLBACK_PORT
- except ValueError:
- return FALLBACK_PORT
-
-
-def _url_with_port(url: str, port: int) -> str:
- parts = urllib.parse.urlsplit(url)
- host = parts.hostname or "127.0.0.1"
- return urllib.parse.urlunsplit(
- (parts.scheme or "http", f"{host}:{port}", parts.path, "", ""))
-
-
-def update_config_api_url_port(port: int, config_path: Optional[Path] = None) -> bool:
- """Rewrite the port inside AUDIOCPP_API_URL in converter/config.py.
-
- Only the quoted URL literal is replaced; surrounding lines and the
- trailing comment are preserved. Returns True when the file was changed.
- """
- path = Path(config_path) if config_path is not None else CONFIG_PATH
- try:
- text = path.read_text(encoding="utf-8")
- except OSError:
- return False
- match = re.search(r'(?m)^(\s*AUDIOCPP_API_URL\s*=\s*")([^"]*)(")', text)
- if not match:
- return False
- new_url = _url_with_port(match.group(2), port)
- if new_url == match.group(2):
- return False
- text = text[:match.start(2)] + new_url + text[match.end(2):]
- try:
- path.write_text(text, encoding="utf-8")
- except OSError:
- return False
- return True
-
-
-def update_config_model_ids(model_id: str,
- clone_model_id: Optional[str] = None,
- config_path: Optional[Path] = None) -> bool:
- """Rewrite AUDIOCPP_MODEL_ID (and AUDIOCPP_CLONE_MODEL_ID when given).
-
- Only the quoted id literals are replaced; surrounding lines and
- comments are preserved. Returns True when the file was changed.
- """
- path = Path(config_path) if config_path is not None else CONFIG_PATH
- try:
- text = path.read_text(encoding="utf-8")
- except OSError:
- return False
- updates: List[Tuple[str, str]] = [("AUDIOCPP_MODEL_ID", model_id)]
- if clone_model_id is not None:
- updates.append(("AUDIOCPP_CLONE_MODEL_ID", clone_model_id))
- changed = False
- for name, value in updates:
- match = re.search(r'(?m)^(\s*' + name + r'\s*=\s*")([^"]*)(")', text)
- if match and match.group(2) != value:
- text = text[:match.start(2)] + value + text[match.end(2):]
- changed = True
- if not changed:
- return False
- try:
- path.write_text(text, encoding="utf-8")
- except OSError:
- return False
- return True
-
-
-def default_model_id(family: str) -> str:
- """Derive a default server entry id from a family name."""
- if family in PREFERRED_IDS:
- return PREFERRED_IDS[family]
- name = family
- if name.endswith("_tts"):
- name = name[:-4]
- return name.replace("_", "") or family
-
-
-def detect_audiocpp_dir() -> Optional[Path]:
- """Best-effort location of a local audio.cpp checkout with model_specs.
-
- Checks the AUDIOCPP_DIR environment variable, then an ``audio.cpp``
- directory in or above the current working directory. Returns the path
- only when it contains a ``model_specs`` directory.
- """
- candidates: List[Path] = []
- env_dir = os.environ.get("AUDIOCPP_DIR")
- if env_dir:
- candidates.append(Path(os.path.expanduser(env_dir)))
- cwd = Path.cwd()
- candidates.append(cwd / "audio.cpp")
- candidates.append(cwd.parent / "audio.cpp")
- candidates.append(cwd.parent.parent / "audio.cpp")
- for candidate in candidates:
- try:
- resolved = candidate.resolve()
- except OSError:
- continue
- if (resolved / "model_specs").is_dir():
- return resolved
- return None
-
-
-# audio.cpp build directories are named ``<platform>-<backend>-<type>`` (e.g.
-# ``linux-cuda-release``, ``windows-vulkan-debug``, ``macos-metal-release``)
-# and the built server lands in ``<that>/bin/audiocpp_server``. The Metal
-# macOS backend is reported as "cpu" here since it is not a separate
-# --backend choice for audiocpp_server.
-_BACKEND_TOKEN_RE = re.compile(r"-(cuda|vulkan|hip|cpu|metal)(?:-|$)")
-
-
-def detect_backend(audiocpp_dir: Path) -> Optional[str]:
- """Best-effort detection of the backend audiocpp_server was built for.
-
- Scans ``audiocpp_dir/build/*`` for build directories that contain a
- built ``bin/audiocpp_server`` (``.exe`` allowed on Windows) and reads
- the backend token out of the directory name (``-cuda-``, ``-vulkan-``,
- ``-hip-`` or ``-cpu-``; ``-metal-`` is mapped to ``cpu``). Returns the
- backend only when exactly one distinct backend was built, so a checkout
- with builds for several backends does not silently pick one. Returns
- None when there is no ``build/`` directory, no built server, or more
- than one distinct backend.
- """
- build_root = audiocpp_dir / "build"
- if not build_root.is_dir():
- return None
- backends: Set[str] = set()
- try:
- build_dirs = sorted(build_root.iterdir(),
- key=lambda p: p.name.lower())
- except OSError:
- return None
- for build_dir in build_dirs:
- if not build_dir.is_dir():
- continue
- server = build_dir / "bin" / "audiocpp_server"
- if not server.exists():
- server_exe = build_dir / "bin" / "audiocpp_server.exe"
- if not server_exe.exists():
- continue
- match = _BACKEND_TOKEN_RE.search(build_dir.name.lower())
- if not match:
- continue
- token = match.group(1)
- backends.add("cpu" if token == "metal" else token)
- if len(backends) == 1:
- return next(iter(backends))
- return None
-
-
-def _default_package(packages: List[dict]) -> Optional[dict]:
- """Pick the default package from a list of packages.
-
- Prefers the package flagged ``default: true``, then the first GGUF
- package, then the first package overall. Returns None for an empty list.
- """
- if not packages:
- return None
- for package in packages:
- if package.get("default"):
- return package
- for package in packages:
- if package.get("format") == "gguf":
- return package
- return packages[0]
-
-
-def load_model_catalog(audiocpp_dir: Path) -> List[dict]:
- """Read model_specs/*.json and return the TTS-capable families.
-
- Each returned entry has: family, display_name, description, languages,
- clone_capable, packages (the full list from the spec), install_id
- (recommended package id), default_path (``models/<target_directory>``),
- and preferred_id. All families are treated equally and listed in
- alphabetical order by display name.
- """
- specs_dir = audiocpp_dir / "model_specs"
- if not specs_dir.is_dir():
- raise NotADirectoryError(
- f"{audiocpp_dir} has no model_specs/ directory; point "
- "--audiocpp-dir at an audio.cpp checkout")
- entries: List[dict] = []
- for spec_path in sorted(specs_dir.glob("*.json")):
- try:
- spec = json.loads(spec_path.read_text(encoding="utf-8"))
- except (OSError, ValueError):
- continue
- tasks = spec.get("tasks") or []
- if "tts" not in tasks and spec.get("category") != "tts":
- continue
- family = spec.get("family") or spec_path.stem
- packages = spec.get("packages") or []
- package = _default_package(packages)
- if package is None:
- # No installable package: skip (cannot be hosted from a path).
- continue
- target_directory = package.get("target_directory") or family
- languages = spec.get("languages") or []
- display_name = spec.get("display_name") or family
- description = spec.get("description") or ""
- entries.append({
- "family": family,
- "display_name": display_name,
- "description": description,
- "languages": languages,
- "tasks": list(tasks),
- "clone_capable": "clone" in tasks,
- "packages": packages,
- "install_id": package.get("id") or family,
- "default_path": f"models/{target_directory}",
- "preferred_id": default_model_id(family),
- })
-
- # All families are treated equally: alphabetical by display name.
- entries.sort(key=lambda entry: entry["display_name"].lower())
- return entries
-
-
-def is_design_package(package: dict) -> bool:
- """Return True when a package's name marks it a voice-design model.
-
- audio.cpp voice-design packages (whose id, display name, or target
- directory mentions "voice design") are the only packages that must be
- hosted with task "vdes"; their role is not in the schema, only in those
- strings, so it is detected from them.
- """
- text = " ".join(str(package.get(key, ""))
- for key in ("id", "display_name", "target_directory"))
- return bool(DESIGN_PACKAGE_RE.search(text))
-
-
-def package_dir_options(entry: dict) -> List[dict]:
- """Return one option per distinct target_directory of a family's packages.
-
- Each option is a dict with: target_directory, install_id (the recommended
- package id inside that directory), design (voice-design package flag), and
- recommended (whether it holds the family's default package). Precisions
- that share a directory (q8_0/bf16/...) collapse to a single option.
- """
- packages = entry.get("packages") or []
- default_pkg = _default_package(packages)
- default_dir = (default_pkg or {}).get("target_directory") or entry["family"]
- by_dir: Dict[str, List[dict]] = {}
- order: List[str] = []
- for package in packages:
- directory = package.get("target_directory") or entry["family"]
- if directory not in by_dir:
- by_dir[directory] = []
- order.append(directory)
- by_dir[directory].append(package)
- options: List[dict] = []
- for directory in order:
- package = _default_package(by_dir[directory])
- options.append({
- "target_directory": directory,
- "install_id": (package or {}).get("id") or directory,
- "design": is_design_package(package or {}),
- "recommended": directory == default_dir,
- })
- # Put the recommended package first for a friendlier checklist.
- options.sort(key=lambda opt: not opt["recommended"])
- return options
-
-
-def ask_package_dirs(entry: dict) -> List[dict]:
- """Choose which of a family's packages to host (multi-select checklist).
-
- Enter selects the recommended package only, matching the default flow.
- """
- options = package_dir_options(entry)
- if len(options) <= 1:
- return options
- default = {opt["target_directory"] for opt in options if opt["recommended"]}
- labels = []
- for opt in options:
- marker = " [recommended]" if opt["recommended"] else ""
- labels.append((f"{opt['install_id']} -> {opt['target_directory']}{marker}",
- opt["target_directory"]))
- chosen = ask_checklist(
- f"Which {entry['display_name']} packages should the server host?",
- labels, default=default)
- return [opt for opt in options if opt["target_directory"] in chosen]
-
-
-def ask_package_task(install_id: str) -> str:
- """Ask how to host a voice-design package: vdes or tts."""
- return ask_menu(
- f"How should the '{install_id}' package be hosted?",
- [
- ("design (vdes) - describe the voice with --instructions",
- TASK_VDES),
- ("tts - normal synthesis", TASK_TTS),
- ],
- default_index=1)
-
-
-def ask_families(catalog: List[dict]) -> List[str]:
- """Show a numbered table and return the chosen family keys.
-
- Input is comma/space-separated numbers; Enter alone selects the first
- entry. At least one family is required.
- """
- rows: List[Tuple[str, str]] = []
- for entry in catalog:
- capabilities = ["tts"]
- if "clone" in entry["tasks"]:
- capabilities.append("cloning")
- if "design" in entry["tasks"]:
- capabilities.append("design")
- name = entry["display_name"]
- if name != entry["family"]:
- name = f"{name} ({entry['family']})"
- rows.append((name, ", ".join(capabilities)))
- number_width = len(str(len(rows)))
- name_width = max([len("Model family")] + [len(name) for name, _ in rows])
- tasks_width = max([len("Tasks")] + [len(tasks) for _, tasks in rows])
- header = (f"{'#'.ljust(number_width)} | "
- f"{'Model family'.ljust(name_width)} | "
- f"{'Tasks'.ljust(tasks_width)}")
- divider = (f"{'-' * number_width}-+-"
- f"{'-' * name_width}-+-"
- f"{'-' * tasks_width}")
- print("Select TTS model families to host (comma-separated numbers,")
- print("or press Enter for the first family):")
- print(header)
- print(divider)
- for number, (name, tasks) in enumerate(rows, 1):
- print(f"{str(number).ljust(number_width)} | "
- f"{name.ljust(name_width)} | "
- f"{tasks.ljust(tasks_width)}")
- while True:
- try:
- answer = input("Choice [1]: ").strip()
- except EOFError:
- return [catalog[0]["family"]]
- if not answer:
- return [catalog[0]["family"]]
- parts = [p for p in re.split(r"[,\s]+", answer) if p]
- indices: List[int] = []
- valid = True
- for part in parts:
- if part.isdigit() and 1 <= int(part) <= len(catalog):
- indices.append(int(part))
- else:
- valid = False
- break
- if valid and indices:
- chosen: List[str] = []
- seen = set()
- for index in indices:
- family = catalog[index - 1]["family"]
- if family not in seen:
- seen.add(family)
- chosen.append(family)
- return chosen
- print(f"Please enter comma-separated numbers between 1 and {len(catalog)}.")
-
-
-def build_model_entry(family: str, model_id: str, model_path: str,
- task: str = TASK_TTS) -> dict:
- """Assemble one server.json model entry.
-
- ``task`` defaults to "tts"; voice design packages are hosted with
- "vdes" so the server runs its design session for speech requests
- (audiobook.py then requires --instructions with that entry).
- """
- return {
- "id": model_id,
- "family": family,
- "path": model_path,
- "task": task,
- "mode": "offline",
- }
-
-
-def build_server_config(host: str, port: int, backend: str, lazy_load: bool,
- model_entries: List[dict],
- voice_dir: Optional[str] = None) -> dict:
- """Assemble the server.json document.
-
- ``voice_dir`` is a server-level cloning voice library; when set, every
- hosted clone-capable family can use its voices with ``--voice``.
- """
- config_doc = {
- "host": host,
- "port": port,
- "backend": backend,
- "lazy_load": lazy_load,
- "models": model_entries,
- }
- if voice_dir:
- config_doc["voice_dir"] = voice_dir
- return config_doc
-
-
-def transcribe_wav_dir(wav_files: list, whisper_model: str) -> Dict[str, str]:
- """Transcribe each wav file and return a mapping of stem -> transcript."""
- transcripts: Dict[str, str] = {}
- for wav_file in wav_files:
- name = wav_file.stem
- print(f"[INFO] Transcribing {wav_file.name} (voice '{name}')...")
- text = transcribe_reference_audio(str(wav_file), model_name=whisper_model)
- if text:
- print(f"[OK] {name}: {text}")
- else:
- print(f"[WARNING] No transcript for '{name}'; cloning works best "
- "with an accurate transcript — consider editing prompt_text "
- "by hand before starting the server")
- transcripts[name] = text or ""
- return transcripts
-
-
-def read_prompt_text(prompt_path: Path) -> Dict[str, str]:
- """Parse a prompt_text file into a stem -> transcript mapping.
-
- Lines are ``<name>|<transcript>``; blank lines are skipped and a line
- without a ``|`` separator is treated as a name with an empty transcript.
- Returns an empty mapping when the file does not exist.
- """
- if not prompt_path.exists():
- return {}
- mapping: Dict[str, str] = {}
- for line in prompt_path.read_text(encoding="utf-8").splitlines():
- if not line.strip():
- continue
- if "|" in line:
- name, _, text = line.partition("|")
- else:
- name, text = line, ""
- mapping[name.strip()] = text
- return mapping
-
-
-def write_prompt_text(wav_dir: Path,
- transcripts: Dict[str, str]) -> Path:
- """Write the voice_dir prompt_text mapping into WAV_DIR.
-
- One ``<basename-without-extension>|<transcript>`` line per voice.
- Returns the path of the written file.
- """
- prompt_path = wav_dir / PROMPT_TEXT_FILENAME
- lines = [f"{name}|{text}" for name, text in transcripts.items()]
- prompt_path.write_text("\n".join(lines) + "\n", encoding="utf-8")
- return prompt_path
-
-
-def print_empty_transcript_warning(transcripts: Dict[str, str]) -> None:
- """Print a loud, final warning for voices whose transcript is empty."""
- empty = sorted(name for name, text in transcripts.items() if not text)
- if not empty:
- return
- bar = "=" * 70
- print()
- print(bar)
- print("[WARNING] MANUAL TRANSCRIPTION REQUIRED")
- print(bar)
- listing = " - " + "\n - ".join(empty) if len(empty) > 1 else f" - {empty[0]}"
- print(f"The following voice(s) have an EMPTY transcript in prompt_text:\n"
- f"{listing}")
- print("Those voices will NOT work until you add an accurate transcript.")
- print(f"Edit {PROMPT_TEXT_FILENAME} in your voice directory and fill in the "
- "text after '|' for each voice above.")
- print(bar)
-
-
-def _apply_port_sync(port: int, accepted: bool) -> None:
- """Write the port into converter/config.py, or report when declined."""
- if accepted:
- if not update_config_api_url_port(port):
- print(f"[WARNING] Could not update {CONFIG_PATH}; edit "
- "AUDIOCPP_API_URL by hand so audiobook.py uses the "
- "new port")
- else:
- print("[WARNING] Left AUDIOCPP_API_URL unchanged; audiobook.py "
- f"will still use port {config_port()}")
-
-
-def _ask_host_port_backend_lazy(args: argparse.Namespace,
- default_lazy: bool,
- detected_backend: Optional[str] = None
- ) -> Tuple[str, int, str, bool]:
- """Ask for (or take from flags) the shared server settings.
-
- DETECTED_BACKEND (from detect_backend) is offered as the default backend
- selection when --backend is not given.
- """
- host = args.host if args.host else ask("Bind host", DEFAULT_HOST)
- port = args.port if args.port is not None else ask_port(config_port())
- if port != config_port():
- if ask_bool(f"Update AUDIOCPP_API_URL in converter/config.py to port "
- f"{port} so audiobook.py talks to this server", True):
- _apply_port_sync(port, True)
- else:
- _apply_port_sync(port, False)
- backend = args.backend if args.backend else \
- ask_backend(detected_backend)
- lazy_load = args.lazy_load or ask_bool(
- "Load models lazily (on first use instead of at startup)", default_lazy)
- return host, port, backend, lazy_load
-
-
-def _decide_transcription(wav_files: list, existing: Dict[str, str],
- prompt_exists: bool, force: bool,
- confirm: Callable[[str, bool], bool]) -> dict:
- """Decide which voices to transcribe; CONFIRM asks the plan questions.
-
- Returns a plan dict: {"mode": "all"|"missing"|"keep", "missing":
- [...], "existing": {...}} — "existing" carries the prompt_text
- mapping read while deciding, so the caller can reuse it instead of
- reading the file again.
- """
- mode = "all"
- missing: List[Path] = []
- if prompt_exists and not force:
- missing = [wav for wav in wav_files
- if not existing.get(wav.stem, "").strip()]
- if not missing:
- if confirm("All voices already transcribed in prompt_text. "
- "Re-transcribe anyway?", False):
- mode = "all"
- else:
- mode = "keep"
- elif confirm("Existing transcription and new .wavs detected, "
- "only transcribe new voices?", True):
- mode = "missing"
- else:
- mode = "all"
- return {"mode": mode, "missing": missing, "existing": existing}
-
-
-def _transcribe(args: argparse.Namespace, include_clone: bool,
- plan: Optional[dict] = None
- ) -> Tuple[Dict[str, str], bool]:
- """Transcribe the wav directory into a stem -> transcript mapping.
-
- Returns the mapping and a flag indicating whether it should be written to
- prompt_text (False when an existing, complete prompt_text is kept as-is).
- When PLAN is given (pre-collected by the TUI) no further questions are
- asked and the prompt_text mapping it already read is reused; otherwise
- the plan is decided with the line prompts.
- """
- if not include_clone:
- print(f"[WARNING] Ignoring {args.input_dir}: no clone-capable family "
- "selected, so voice presets are not used")
- return {}, False
-
- wav_files = find_wav_files(args.input_dir)
- if not wav_files:
- print(f"[WARNING] No .wav files found in {args.input_dir}; writing the "
- "config without a voice_dir")
- return {}, False
-
- prompt_path = args.input_dir / PROMPT_TEXT_FILENAME
- if plan is None:
- existing = read_prompt_text(prompt_path) if (
- prompt_path.exists() and not args.force) else {}
- plan = _decide_transcription(
- wav_files, existing, prompt_path.exists(), args.force,
- lambda question, default: ask_bool(question, default))
- else:
- existing = plan.get("existing") or {}
-
- if plan["mode"] == "keep":
- print(f"[INFO] Kept existing {prompt_path}; all voices were "
- "already transcribed, nothing new to transcribe")
- return existing, False
-
- if whisper_backend_available() is None:
- print("[WARNING] Neither faster_whisper nor whisper was found, so "
- "reference .wav files cannot be transcribed automatically and "
- "every transcript will be empty.")
- print(" Install whisper (or faster_whisper) in your "
- "audiobook environment to transcribe automatically; otherwise "
- "transcripts must be added by hand (see the warning at the end).")
-
- if plan["mode"] == "missing":
- new_transcripts = transcribe_wav_dir(plan["missing"], args.whisper_model)
- transcripts = dict(existing)
- transcripts.update(new_transcripts)
- else:
- transcripts = transcribe_wav_dir(wav_files, args.whisper_model)
- return transcripts, True
-
-
-def _offer_config_model_id_sync(model_id: str,
- accepted: Optional[bool] = None) -> None:
- """Offer to point converter/config.py at a single hosted model entry.
-
- The converter requests the model id configured in AUDIOCPP_MODEL_ID,
- and single-model servers use the same id for the clone entry, so both
- ids are rewritten together. When ACCEPTED is None the user is asked
- (line prompt); otherwise the given decision is applied.
- """
- if config.AUDIOCPP_MODEL_ID == model_id \
- and config.AUDIOCPP_CLONE_MODEL_ID == model_id:
- return
- if accepted is None:
- accepted = ask_bool("Update AUDIOCPP_MODEL_ID and AUDIOCPP_CLONE_MODEL_ID "
- f"in converter/config.py to '{model_id}' so "
- "audiobook.py uses this model", True)
- if accepted:
- if not update_config_model_ids(model_id, model_id):
- print(f"[WARNING] Could not update {CONFIG_PATH}; edit "
- "AUDIOCPP_MODEL_ID and AUDIOCPP_CLONE_MODEL_ID by hand so "
- "audiobook.py uses this model")
- else:
- print("[WARNING] Left the model ids unchanged; audiobook.py will "
- f"still request model '{config.AUDIOCPP_MODEL_ID}'")
-
-
-def _build_entries(family_keys: List[str], chosen: Dict[str, List[dict]],
- catalog_by_family: Dict[str, dict],
- task_picker: Callable[[str], str],
- id_picker: Callable[[str, str, str], str]
- ) -> Tuple[List[dict], List[str], List[Tuple[str, str]],
- List[str], bool]:
- """Build server.json model entries from the selected families/packages.
-
- TASK_PICKER is called for each design package to choose vdes/tts;
- ID_PICKER resolves a duplicate server entry id. Returns (model_entries,
- entry_ids, install_guidance, design_entry_ids, include_clone).
- """
- model_entries: List[dict] = []
- entry_ids: List[str] = []
- install_guidance: List[Tuple[str, str]] = []
- design_entry_ids: List[str] = []
- include_clone = False
- for family in family_keys:
- entry = catalog_by_family[family]
- include_clone = include_clone or entry["clone_capable"]
- for opt in chosen[family]:
- task = task_picker(opt["install_id"]) if opt["design"] else TASK_TTS
- base_id = (f"{entry['preferred_id']}-design"
- if task == TASK_VDES else entry["preferred_id"])
- model_id = base_id
- if model_id in entry_ids:
- model_id = id_picker(entry["display_name"], opt["install_id"],
- f"{base_id}-2")
- entry_ids.append(model_id)
- model_entries.append(build_model_entry(
- family, model_id, f"models/{opt['target_directory']}",
- task=task))
- install_guidance.append((entry["display_name"], opt["install_id"]))
- if task == TASK_VDES:
- design_entry_ids.append(model_id)
- return (model_entries, entry_ids, install_guidance,
- design_entry_ids, include_clone)
-
-
-def _write_and_advise(audiocpp_dir: Path, wav_dir: Optional[Path],
- output_path: Path, model_entries: List[dict],
- install_guidance: List[Tuple[str, str]], host: str,
- port: int, backend: str, lazy_load: bool,
- transcripts: Dict[str, str], write_prompt: bool) -> None:
- """Console phase shared by both UI modes: write files, print summary.
-
- After a successful run the console output is the path of the written
- server.json. The model install commands (and optional automatic
- download) are handled separately by _install_models, called by both
- UI modes once the user has decided whether to download.
- """
- voice_dir: Optional[str] = None
- if transcripts:
- if write_prompt:
- prompt_path = wav_dir / PROMPT_TEXT_FILENAME
- write_prompt_text(wav_dir, transcripts)
- print(f"[OK] Wrote {prompt_path}")
- voice_dir = str(wav_dir.resolve())
-
- server_config = build_server_config(
- host=host, port=port, backend=backend, lazy_load=lazy_load,
- model_entries=model_entries, voice_dir=voice_dir)
-
- with output_path.open("w", encoding="utf-8") as handle:
- json.dump(server_config, handle, indent=2, ensure_ascii=False)
- handle.write("\n")
-
- count = len(model_entries)
- print(f"Wrote {output_path.resolve()} with {count} "
- f"{'entry' if count == 1 else 'entries'}.")
-
-
-def _install_models(audiocpp_dir: Path,
- install_guidance: List[Tuple[str, str]],
- download: bool) -> None:
- """Print and optionally run the model install commands.
-
- One ``python <manager> install <id>`` command per hosted model (de-duped
- by install id). When DOWNLOAD is True each command is run in the audio.cpp
- checkout via ``subprocess.run`` so the models are downloaded automatically;
- a failing install is reported as a warning and does not abort the remaining
- downloads. When DOWNLOAD is False (or the model manager is missing) the
- commands are only printed, copy-pasteable as before.
- """
- manager = audiocpp_dir / "tools" / "model_manager_v2.py"
- seen: Set[str] = set()
- install_ids: List[str] = []
- for _, install_id in install_guidance:
- if install_id not in seen:
- seen.add(install_id)
- install_ids.append(install_id)
-
- if download and not manager.is_file():
- print(f"[WARNING] {manager} not found; printing the install commands "
- "instead of running them")
- download = False
-
- for install_id in install_ids:
- command = f"python {manager} install {install_id}"
- if not download:
- print(command)
- continue
- print(f"[INFO] Downloading {install_id}...")
- try:
- result = subprocess.run(
- [sys.executable, str(manager), "install", install_id],
- cwd=str(audiocpp_dir))
- except OSError as exc:
- print(f"[WARNING] Could not run {command}: {exc}")
- continue
- if result.returncode != 0:
- print(f"[WARNING] install {install_id} exited with code "
- f"{result.returncode}; the model may need to be downloaded "
- "by hand")
-
-
-def _decide_download(audiocpp_dir: Path,
- confirm: Callable[[str, bool], bool]) -> bool:
- """Ask whether to download the selected models now.
-
- CONFIRM asks the yes/no question (ask_bool for the line prompts, a TUI
- confirm for the wizard). When the audio.cpp model manager is missing the
- prompt is skipped and False is returned, so the install commands are only
- printed rather than offered to run.
- """
- manager = audiocpp_dir / "tools" / "model_manager_v2.py"
- if not manager.is_file():
- return False
- return confirm(
- "Automatically download the selected models with model_manager_v2.py "
- "now?", False)
-
-
-def _build_tree_families(catalog: List[dict]) -> List[dict]:
- """Shape the catalog into the checkbox_tree widget's family list."""
- families: List[dict] = []
- for entry in catalog:
- capabilities = ["tts"]
- if "clone" in entry["tasks"]:
- capabilities.append("cloning")
- if "design" in entry["tasks"]:
- capabilities.append("design")
- name = entry["display_name"]
- if name != entry["family"]:
- name = f"{name} ({entry['family']})"
- options = []
- for opt in package_dir_options(entry):
- options.append({
- "key": opt["target_directory"],
- "label": opt["install_id"],
- "recommended": opt["recommended"],
- })
- families.append({
- "label": name,
- "detail": ", ".join(capabilities),
- "options": options,
- })
- return families
-
-
-def _wizard(stdscr, args: argparse.Namespace, parser: argparse.ArgumentParser
- ) -> Optional[dict]:
- """Run every TUI screen; return the collected settings, or None to abort.
-
- The wizard is a step state machine; each screen group is one step, and
- Esc anywhere but the first step falls back to the previous group (the
- widget returns the _GO_BACK sentinel, or a callback raises _GoBack). On
- the first screen (the audio.cpp checkout browser) Esc aborts the whole
- wizard as before.
- """
- tui = _load_tui()
-
- def ask_confirm(question: str, default: bool) -> bool:
- result = tui.confirm(stdscr, question, default=default,
- cancel_value=_GO_BACK)
- if result is _GO_BACK:
- raise _GoBack()
- return result
-
- step = 0
- while True:
- if step == 0:
- # Checkout browser + the output path/overwrite confirmation. The
- # browser asks for the checkout root and finds model_specs/ inside
- # it (picking the model_specs directory itself works too — its
- # parent is used). A highlighted subdirectory named "audio.cpp"
- # that already contains model_specs/ is auto-accepted on
- # Enter/Right, skipping the "[ Use this directory ]" step.
- # Pressing Esc on an overwrite confirmation returns here instead
- # of aborting: the browser then restarts inside the previously
- # accepted checkout with auto-accept disabled, so a wrong guess
- # can be corrected. An explicit --audiocpp-dir flag has no
- # browser to return to, so Esc still aborts there. Esc on the
- # browser itself is the first step, so it aborts the wizard.
- auto_accept = True
- browser_start: Path = Path.cwd()
- force_browse = False
- while True:
- audiocpp_dir = args.audiocpp_dir
- if audiocpp_dir is None:
- audiocpp_dir = detect_audiocpp_dir()
- if force_browse:
- audiocpp_dir = None
- if audiocpp_dir is None:
- audiocpp_dir = tui.browse_directory(
- stdscr, "Select your audio.cpp directory",
- validate=lambda p: None if _resolve_audiocpp_root(p)
- else "No model_specs/ directory here",
- info=_audiocpp_root_status,
- preview=_audiocpp_root_preview,
- help_lines=["The root folder of your audio.cpp "
- "checkout;",
- "it is the one that contains "
- "model_specs/"],
- start=browser_start,
- auto_select=_checkout_auto_select if auto_accept
- else None)
- audiocpp_dir = Path(audiocpp_dir).resolve()
- if not audiocpp_dir.is_dir():
- raise _TuiError(f"audio.cpp checkout not found: "
- f"{audiocpp_dir}")
- root = _resolve_audiocpp_root(audiocpp_dir)
- if root is None:
- raise _TuiError(
- f"{audiocpp_dir} has no model_specs/ directory; "
- "select the root of your audio.cpp checkout")
- audiocpp_dir = root
- try:
- catalog = load_model_catalog(audiocpp_dir)
- except NotADirectoryError as exc:
- raise _TuiError(str(exc))
- if not catalog:
- raise _TuiError(f"No TTS model families found in "
- f"{audiocpp_dir}/model_specs; check the "
- "checkout is up to date")
- catalog_by_family = {entry["family"]: entry
- for entry in catalog}
-
- output_path = args.output if args.output is not None \
- else audiocpp_dir / "server.json"
- esc_back = args.audiocpp_dir is None
- went_back = False
- if not args.force and output_path.exists():
- decision = tui.confirm(
- stdscr, f"{output_path} already exists. Overwrite?",
- default=True,
- cancel_value=_GO_BACK if esc_back else None)
- if decision is _GO_BACK:
- went_back = True
- elif decision is False:
- if args.output is None:
- output_path = Path.cwd() / "server.json"
- if output_path.exists():
- decision = tui.confirm(
- stdscr,
- f"{output_path} already exists. "
- "Overwrite?",
- default=True,
- cancel_value=_GO_BACK if esc_back else None)
- if decision is _GO_BACK:
- went_back = True
- elif decision is False:
- return None
- else:
- return None
- if went_back:
- auto_accept = False
- browser_start = audiocpp_dir
- force_browse = True
- continue
- break
- detected_backend = detect_backend(audiocpp_dir)
- step = 1
- continue
-
- if step == 1:
- # Families and packages (flag or tree). Esc returns to the
- # checkout browser (step 0).
- chosen: Dict[str, List[dict]] = {}
- if args.families is not None:
- requested = [f.strip() for f in args.families.split(",")
- if f.strip()]
- unknown = [f for f in requested if f not in catalog_by_family]
- if unknown:
- raise _TuiError(
- f"Unknown family in --families: {', '.join(unknown)}. "
- f"Available: {', '.join(catalog_by_family)}")
- family_keys: List[str] = []
- for family in requested:
- if family not in family_keys:
- family_keys.append(family)
- chosen[family] = [opt for opt in package_dir_options(
- catalog_by_family[family]) if opt["recommended"]]
- else:
- tree_families = _build_tree_families(catalog)
- picked = tui.checkbox_tree(
- stdscr, "Select TTS model families to host",
- tree_families, expand_all=args.all_packages,
- back_value=_GO_BACK)
- if picked is _GO_BACK:
- step = 0
- continue
- family_keys = []
- for family_index, option_key in picked:
- family = catalog[family_index]["family"]
- if family not in chosen:
- chosen[family] = []
- family_keys.append(family)
- chosen[family].append(option_key)
- for family in list(chosen):
- keyed = {opt["target_directory"]: opt
- for opt in package_dir_options(
- catalog_by_family[family])}
- chosen[family] = [keyed[key] for key in chosen[family]]
- step = 2
- continue
-
- if step == 2:
- # Design task menus and duplicate-id renames. Esc anywhere here
- # falls back to the families tree (step 1).
- def task_picker(install_id: str) -> str:
- result = tui.menu(
- stdscr,
- f"How should the '{install_id}' package be hosted?",
- [
- ("design (vdes) - describe the voice with "
- "--instructions", TASK_VDES),
- ("tts - normal synthesis", TASK_TTS),
- ], default_index=0, back_value=_GO_BACK)
- if result is _GO_BACK:
- raise _GoBack()
- return result
-
- def id_picker(display_name: str, install_id: str,
- default: str) -> str:
- result = tui.line_edit(
- stdscr,
- f"Server model id for {display_name} package "
- f"'{install_id}'", default, back_value=_GO_BACK)
- if result is _GO_BACK:
- raise _GoBack()
- return result
-
- try:
- model_entries, entry_ids, install_guidance, \
- design_entry_ids, include_clone = _build_entries(
- family_keys, chosen, catalog_by_family,
- task_picker, id_picker)
- except _GoBack:
- step = 1
- continue
- step = 3
- continue
-
- if step == 3:
- # Server settings (host, port, port-sync, backend, lazy). Esc on
- # any of them falls back to the previous group (step 2).
- if args.host:
- host = args.host
- else:
- host = tui.line_edit(
- stdscr, "Bind host", DEFAULT_HOST,
- help_lines=["The IP address audiocpp will be hosted on",
- "127.0.0.1 (this machine) is probably "
- "correct"], back_value=_GO_BACK)
- if host is _GO_BACK:
- step = 2
- continue
- if args.port is not None:
- port = args.port
- else:
- port_text = tui.line_edit(
- stdscr, "Port", str(config_port()),
- validate=lambda s: None if (s.isdigit()
- and 1 <= int(s) <= 65535)
- else "Enter a port number between 1 and 65535",
- help_lines=["The port audiocpp will be hosted on"],
- back_value=_GO_BACK)
- if port_text is _GO_BACK:
- step = 2
- continue
- port = int(port_text)
- sync_port: Optional[bool] = None
- if port != config_port():
- sync_port = tui.confirm(
- stdscr, f"Update AUDIOCPP_API_URL in converter/config.py "
- f"to port {port} so audiobook.py talks to this server",
- default=True, cancel_value=_GO_BACK)
- if sync_port is _GO_BACK:
- step = 2
- continue
- if args.backend:
- backend = args.backend
- else:
- backend_options, backend_default = \
- _backend_options(detected_backend)
- backend = tui.menu(
- stdscr, "Which inference backend was audiocpp_server "
- "built for?", backend_options,
- default_index=backend_default, back_value=_GO_BACK)
- if backend is _GO_BACK:
- step = 2
- continue
- default_lazy = len(model_entries) > 1
- if args.lazy_load:
- lazy_load = True
- else:
- lazy_load = tui.confirm(
- stdscr, "Load models lazily (on first use instead of at "
- "startup)", default=default_lazy, cancel_value=_GO_BACK)
- if lazy_load is _GO_BACK:
- step = 2
- continue
- step = 4
- continue
-
- if step == 4:
- # Wav directory (flag, browsed when cloning, else skipped). Esc
- # falls back to the server settings (step 3).
- if args.input_dir is not None:
- wav_dir = args.input_dir
- elif include_clone:
- wav_start = detect_wav_dir(audiocpp_dir, TTS_ROOT)
- wav_dir = tui.browse_directory(
- stdscr, "Select the directory with your .wav voices",
- info=_wav_dir_info, preview=_wav_dir_preview,
- start=wav_start if wav_start is not None else Path.cwd(),
- back_value=_GO_BACK)
- if wav_dir is _GO_BACK:
- step = 3
- continue
- else:
- wav_dir = None
- step = 5
- continue
-
- if step == 5:
- # Transcription plan (questions only; transcription runs after).
- # Esc falls back to the wav browser (step 4).
- plan: Optional[dict] = None
- if include_clone and wav_dir is not None:
- wav_files = find_wav_files(wav_dir)
- if wav_files:
- prompt_path = wav_dir / PROMPT_TEXT_FILENAME
- existing = read_prompt_text(prompt_path) if (
- prompt_path.exists() and not args.force) else {}
- try:
- plan = _decide_transcription(
- wav_files, existing, prompt_path.exists(),
- args.force, ask_confirm)
- except _GoBack:
- step = 4
- continue
- step = 6
- continue
-
- if step == 6:
- # Single-model id sync decision. Esc falls back to the
- # transcription plan (step 5).
- sync_model_ids: Optional[bool] = None
- if len(entry_ids) == 1 and not (
- config.AUDIOCPP_MODEL_ID == entry_ids[0]
- and config.AUDIOCPP_CLONE_MODEL_ID == entry_ids[0]):
- sync_model_ids = tui.confirm(
- stdscr, "Update AUDIOCPP_MODEL_ID and "
- "AUDIOCPP_CLONE_MODEL_ID in converter/config.py to "
- f"'{entry_ids[0]}' so audiobook.py uses this model",
- default=True, cancel_value=_GO_BACK)
- if sync_model_ids is _GO_BACK:
- step = 5
- continue
- step = 8
- continue
-
- if step == 8:
- # Automatic model download (or print the install commands). Esc
- # falls back to the model-id sync (step 6).
- try:
- download = _decide_download(audiocpp_dir, ask_confirm)
- except _GoBack:
- step = 6
- continue
- return {
- "audiocpp_dir": audiocpp_dir,
- "catalog": catalog,
- "catalog_by_family": catalog_by_family,
- "output_path": output_path,
- "family_keys": family_keys,
- "chosen": chosen,
- "model_entries": model_entries,
- "entry_ids": entry_ids,
- "install_guidance": install_guidance,
- "design_entry_ids": design_entry_ids,
- "include_clone": include_clone,
- "host": host,
- "port": port,
- "backend": backend,
- "lazy_load": lazy_load,
- "sync_port": sync_port,
- "sync_model_ids": sync_model_ids,
- "wav_dir": wav_dir,
- "plan": plan,
- "download": download,
- }
-
-
-def _run_tui(args: argparse.Namespace, parser: argparse.ArgumentParser) -> int:
- """Run the TUI wizard, then the shared console phase."""
- import curses
- tui = _load_tui()
- try:
- settings = curses.wrapper(_wizard, args, parser)
- except _TuiError as exc:
- print(f"[ERROR] {exc}", file=sys.stderr)
- return 2
- except tui.WizardCancelled:
- print("\n[INFO] Cancelled; nothing was written")
- return 1
- try:
- curses.curs_set(1) # restore the text cursor hidden by the TUI
- except curses.error:
- pass
- if settings is None:
- print("[INFO] Aborted; existing server.json kept")
- return 1
-
- # Port sync (applied now that the terminal is back).
- if settings["sync_port"] is True:
- _apply_port_sync(settings["port"], True)
- elif settings["sync_port"] is False:
- _apply_port_sync(settings["port"], False)
-
- # Transcription (console; the questions were already answered in the TUI).
- args.input_dir = settings["wav_dir"]
- if settings["include_clone"]:
- transcripts, write_prompt = _transcribe(args, True, plan=settings["plan"])
- elif args.input_dir is not None:
- print(f"[WARNING] Ignoring {args.input_dir}: no clone-capable family "
- "selected, so voice presets are not used")
- transcripts, write_prompt = {}, False
- else:
- transcripts, write_prompt = {}, False
-
- _write_and_advise(
- settings["audiocpp_dir"], settings["wav_dir"], settings["output_path"],
- settings["model_entries"], settings["install_guidance"],
- settings["host"], settings["port"], settings["backend"],
- settings["lazy_load"], transcripts, write_prompt)
-
- if len(settings["entry_ids"]) == 1:
- _offer_config_model_id_sync(settings["entry_ids"][0],
- settings["sync_model_ids"])
- print_empty_transcript_warning(transcripts)
- _install_models(settings["audiocpp_dir"], settings["install_guidance"],
- settings["download"])
- return 0
-
-
-def main() -> int:
- parser = argparse.ArgumentParser(
- description="Generate a server.json for the audio.cpp audiocpp_server "
- "hosting one or more TTS model families used by this converter.")
- parser.add_argument("--wavs", type=resolve_wav_dir_arg, default=None,
- dest="input_dir", metavar="WAV_DIR",
- help="Directory with .wav reference files to publish as "
- "a server-level voice_dir cloning library (asked "
- "for when omitted)")
- parser.add_argument("--output", type=Path, default=None,
- help="Output path for server.json (default: "
- "server.json inside the audio.cpp checkout; if it "
- "already exists you are asked [Y/n] to overwrite, "
- "and answering 'n' writes server.json in the "
- "current directory instead)")
- parser.add_argument("--audiocpp-dir", type=normalize_dir_arg, default=None,
- help="Path to a local audio.cpp checkout containing a "
- "model_specs/ directory (default: detected from "
- "AUDIOCPP_DIR or an audio.cpp directory next to/above "
- "the current working directory; prompted otherwise)")
- parser.add_argument("--families", type=str, default=None,
- help="Comma-separated model families to host, as named "
- "in the audio.cpp catalog (e.g. "
- "qwen3_tts,higgs_audio_tts). Skips the family "
- "checklist")
- parser.add_argument("--all-packages", action="store_true",
- help="Instead of hosting each family's recommended "
- "package, offer a checklist of every installable "
- "package (distinct target_directory) so several "
- "packages of one family can be hosted at once. "
- "In the TUI this pre-expands every family in the "
- "tree (which always lists all packages)")
- parser.add_argument("--host", type=str, default=None,
- help="Bind host for the server (default: 127.0.0.1)")
- parser.add_argument("--port", type=int, default=None,
- help="Port for the server (default: the port in "
- "AUDIOCPP_API_URL from converter/config.py)")
- parser.add_argument("--backend", choices=BACKENDS, default=None,
- help="Inference backend audiocpp_server was built "
- "for (default: auto-detected from the checkout's "
- "build/ directory, else cuda)")
- parser.add_argument("--lazy-load", action="store_true",
- help="Load models on first use instead of at startup "
- "(default: on when more than one model is hosted)")
- parser.add_argument("--whisper-model", type=str, default="base",
- help="Whisper model size for transcription "
- "(default: base)")
- parser.add_argument("--force", action="store_true",
- help="Overwrite the output file (and prompt_text) "
- "without prompting")
- parser.add_argument("--notui", action="store_true",
- help="Use the classic line prompts instead of the "
- "full-screen TUI (automatic when curses is "
- "unavailable or stdin/stdout is not a terminal)")
- args = parser.parse_args()
-
- if args.input_dir is not None and not args.input_dir.is_dir():
- parser.error(
- f"WAV directory not found: {args.input_dir}\n"
- f" (resolved from the current working directory: "
- f"{Path.cwd()})\n"
- " --wavs must be a directory containing the .wav "
- "reference files to use as voice cloning presets")
-
- if _tui_enabled(args):
- return _run_tui(args, parser)
-
- # ---- Line-prompt flow (original behaviour). ---------------------------
-
- # Resolve the wav directory (flag, else prompt). The prompt default is
- # the unique directory that directly contains .wav files across the
- # audio.cpp checkout (best-effort detected here) and the
- # tts-audiobook-generator root, so the user usually just presses Enter.
- if args.input_dir is None:
- tentative_checkout = args.audiocpp_dir or detect_audiocpp_dir()
- wav_start = detect_wav_dir(tentative_checkout, TTS_ROOT) \
- if tentative_checkout is not None else None
- default = str(wav_start) if wav_start is not None else ""
- answer = ask("Directory with .wav reference files", default)
- args.input_dir = resolve_wav_dir_arg(answer) if answer else None
- if args.input_dir is None:
- parser.error("--wavs is required: a directory containing the .wav "
- "reference files to use as voice cloning presets")
- if not args.input_dir.is_dir():
- parser.error(
- f"WAV directory not found: {args.input_dir}\n"
- f" (resolved from the current working directory: "
- f"{Path.cwd()})\n"
- " --wavs must be a directory containing the .wav "
- "reference files to use as voice cloning presets")
-
- # Resolve the audio.cpp checkout and load its model catalog.
- audiocpp_dir = args.audiocpp_dir
- if audiocpp_dir is None:
- audiocpp_dir = detect_audiocpp_dir()
- if audiocpp_dir is None:
- print("[INFO] Could not find an audio.cpp checkout next to or above "
- "the current directory.")
- answer = ask("Path to your audio.cpp checkout", "")
- audiocpp_dir = normalize_dir_arg(answer) if answer else None
- if not audiocpp_dir:
- parser.error(
- "An audio.cpp checkout is required to read the model catalog. "
- "Clone one with `git clone https://github.com/0xShug0/audio.cpp` "
- "and pass --audiocpp-dir PATH (or set the AUDIOCPP_DIR environment "
- "variable)")
- audiocpp_dir = audiocpp_dir.resolve()
- if not audiocpp_dir.is_dir():
- parser.error(f"audio.cpp checkout not found: {audiocpp_dir}")
- root = _resolve_audiocpp_root(audiocpp_dir)
- if root is None:
- parser.error(f"{audiocpp_dir} has no model_specs/ directory; point "
- "--audiocpp-dir at the root of an audio.cpp checkout")
- audiocpp_dir = root
- try:
- catalog = load_model_catalog(audiocpp_dir)
- except NotADirectoryError as exc:
- parser.error(str(exc))
- if not catalog:
- parser.error(
- f"No TTS model families found in {audiocpp_dir}/model_specs; "
- "check the checkout is up to date")
-
- # Resolve the server.json output path. It defaults to the audio.cpp
- # checkout; an existing file is overwritten only with confirmation, and a
- # declined overwrite of the default location falls back to the current
- # working directory.
- output_path = args.output if args.output is not None \
- else audiocpp_dir / "server.json"
- if not args.force and output_path.exists() \
- and not ask_bool(f"{output_path} already exists. Overwrite?", True):
- if args.output is None:
- output_path = Path.cwd() / "server.json"
- if output_path.exists() and not ask_bool(
- f"{output_path} already exists. Overwrite?", True):
- print("[INFO] Aborted; existing server.json kept")
- return 1
- else:
- print("[INFO] Aborted; existing server.json kept")
- return 1
-
- # Select families.
- if args.families is not None:
- requested = [f.strip() for f in args.families.split(",") if f.strip()]
- catalog_families = {entry["family"] for entry in catalog}
- unknown = [f for f in requested if f not in catalog_families]
- if unknown:
- parser.error(
- f"Unknown family in --families: {', '.join(unknown)}. "
- f"Available: {', '.join(entry['family'] for entry in catalog)}")
- family_keys: List[str] = []
- for fam in requested:
- if fam not in family_keys:
- family_keys.append(fam)
- else:
- family_keys = ask_families(catalog)
-
- catalog_by_family = {entry["family"]: entry for entry in catalog}
-
- chosen: Dict[str, List[dict]] = {}
- for family in family_keys:
- entry = catalog_by_family[family]
- if args.all_packages:
- chosen[family] = ask_package_dirs(entry)
- else:
- chosen[family] = [opt for opt in package_dir_options(entry)
- if opt["recommended"]]
-
- model_entries, entry_ids, install_guidance, design_entry_ids, include_clone = \
- _build_entries(family_keys, chosen, catalog_by_family,
- task_picker=lambda install_id: ask_package_task(install_id),
- id_picker=lambda display_name, install_id, base_id: ask(
- f"Server model id for {display_name} package "
- f"'{install_id}'", f"{base_id}-2"))
-
- # Default to lazy loading when hosting more than one model entry: a
- # single-entry server loads at startup, while a multi-entry server avoids
- # loading every model until it is actually used.
- default_lazy = len(model_entries) > 1
- detected_backend = detect_backend(audiocpp_dir)
- host, port, backend, lazy_load = _ask_host_port_backend_lazy(
- args, default_lazy, detected_backend)
-
- transcripts, write_prompt = _transcribe(args, include_clone)
-
- _write_and_advise(
- audiocpp_dir, args.input_dir, output_path, model_entries,
- install_guidance, host, port, backend, lazy_load, transcripts,
- write_prompt)
-
- if len(entry_ids) == 1:
- _offer_config_model_id_sync(entry_ids[0])
- print_empty_transcript_warning(transcripts)
-
- download = _decide_download(
- audiocpp_dir, lambda question, default: ask_bool(question, default))
- _install_models(audiocpp_dir, install_guidance, download)
- return 0
-
-
-if __name__ == "__main__":
- sys.exit(main())