From 610d6828fa8300efa3f8df3d3bf8e3c00e24c1dc Mon Sep 17 00:00:00 2001 From: historia Date: Mon, 24 Aug 2026 20:41:51 -0400 Subject: remove: partial feature to point at existing backend checkout --- README.md | 54 ++++++------------- app/backends/audiocpp.py | 101 ++++++------------------------------ app/backends/faster.py | 35 +++---------- app/backends/qwen.py | 19 ++----- app/tests/test_backends_audiocpp.py | 34 ------------ app/tests/test_tui.py | 71 ------------------------- app/ui/tui.py | 21 ++------ 7 files changed, 46 insertions(+), 289 deletions(-) diff --git a/README.md b/README.md index 7f07fde..88b5342 100644 --- a/README.md +++ b/README.md @@ -13,6 +13,12 @@ The converter sends text extracted from your books to a locally running TTS serv - Automatic metadata (title/artist/album tags, chapter track numbers) and a generated cover - Supports text-to-speech, voice cloning, voice design, and per-model controls like emotion/speed +| Backend | Description | +| -------------------------------------------------------------------- | ------------------------------------------------------ | +| [audio.cpp](https://github.com/0xShug0/audio.cpp) | Newer C++ TTS backend that supports many recent models | +| [Qwen-TTS](https://pypi.org/project/qwen-tts/) | Qwen demo server (qwen-tts-demo) | +| [Faster-Qwen-TTS](https://github.com/andimarafioti/faster-qwen3-tts) | Qwen server with 2-8x faster inference for NVidia GPUs | + ## Prerequisites - Python 3.12+ @@ -20,50 +26,32 @@ The converter sends text extracted from your books to a locally running TTS serv ## Quick Start -Download the project +1. Download the project ```bash git clone https://git.historia.vg/git/tts-audiobook-generator cd tts-audiobook-generator ``` -- `./input` - Put your book files here +2. Put your files in the directories + +- `./input` - Text files to be processed (`epub`, etc.) - `./output` - Audio files will output here -- `./voices` - Put .wav files of voices to clone here (10-20 seconds) +- `./voices` - `.wav` files of voices to clone (10-20 seconds) -Run `audiobook.py`. It will create a venv `./app/envs/tts` and install all requirements. +3. Run `audiobook.py`. It will create a venv `./app/envs/tts` and automatically install all requirements. ``` python audiobook.py ``` +4. When the TUI comes up, go to `Configure Backends > Install Backend`. Install `audio.cpp`, which supports numerous TTS models. It will automatically be cloned and built in the venv (this will take a while). -## Quick start (TUI) - -Run the generator with no arguments in a terminal: - -```bash -python audiobook.py -``` - -A full-screen TUI opens and shows each backend's status in a table — **unavailable** (red, name dimmed: not installed and no server running), **installed** (orange), or **running** (green) with the source(s) in brackets: `[local]` for a server this tool started, `[remote]` for an externally-run server found by probing the backend's remote URL, or `[local, remote]` when both are up. From the menu you can: - -- **Convert books…** — set everything on one screen. The first field picks the **Backend**: each backend appears as a managed entry (e.g. `audio.cpp`) when it's installed and configured here, plus a `[remote]` entry (e.g. `audio.cpp [remote]`) when a running server was found at its remote URL. The rest of the options change to what that backend supports: model, voice and instructions for audio.cpp; speaker or clone .wav for qwen; voice for faster — plus output format, speed, whether to combine all chapters into one file, and debug mode. A managed entry reads its local `server.json` / `voices.json`; a `[remote]` entry queries the server itself instead (audio.cpp lists its models and voices over HTTP, faster asks you to type a voice name). Focus starts on **Generate!**, so Enter accepts the defaults. The "combine chapters" option is hidden for `m4b`, which is always one file. -- After **Generate!**, a full-screen run view takes over instead of dumping you into console output. The top shows the server status — *starting* (a managed server that needed booting is spawned and waited on until it actually answers HTTP, not just accepts TCP connections), *ready*, *processing*, or *error* — and the bottom shows the conversion with a progress bar for the current book's chunks (`Chunk 45/120`) and elapsed time. Esc or `q` first asks whether to cancel processing, then (when this run started the server) whether to shut it down, then returns to the menu. An error — the server exits while booting, stops mid-conversion, or a chunk fails and the book aborts — switches the corresponding state to *error* and waits for a key before returning to the menu, so the failure is never scrolled away (full detail stays in `app/logs/audiobook_.log`), or -- **Configure backends…** — one menu for installing, configuring, and removing backends. Its options are populated from what's currently detected: **Install Backend** (when any backend isn't installed yet), **Configure audio.cpp / qwen-tts / faster-qwen3-tts** (one per installed backend — rerunning its setup wizard acts as a "modify": an existing `server.json` / `voices.json` is loaded and its values pre-filled instead of being overwritten, and audio.cpp offers to delete already-downloaded models you uncheck), **Download Missing Models (audio.cpp)** (runs `model_manager_v2.py` for every model in `server.json` that isn't downloaded yet), and **Uninstall Backend** (removes a backend's files from the managed venv), or -- **Server…** — manually start or stop a configured backend's server (the hub spawns it in the managed venv and polls until it answers). +5. -**Server…** only appears once at least one backend is installed — a merely-running external server unlocks **Convert books…**, but starting/stopping its server needs it on this machine. **Configure backends…** is always available (there is always something to install or remove). +## CLI Options -Everything the TUI does can also be scripted with flags: `python audiobook.py --backend audiocpp --model higgs --voice narrator`, or `python app/backends/audiocpp.py --families higgs_audio_tts --clone --build-backend cuda`. - -You need one of the following backends (the TUI sets them up for you; manual steps below): - -| Backend | Description | -| -------------------------------------------------------------------- | ------------------------------------------------------ | -| [audio.cpp](https://github.com/0xShug0/audio.cpp) | Newer C++ TTS backend that supports many recent models | -| [Qwen-TTS](https://pypi.org/project/qwen-tts/) | Qwen demo server (qwen-tts-demo) | -| [Faster-Qwen-TTS](https://github.com/andimarafioti/faster-qwen3-tts) | Qwen server with 2-8x faster inference for NVidia GPUs | +Everything the TUI does can also be scripted with flags: `python audiobook.py --backend audiocpp --model higgs --voice narrator`. ## Options @@ -104,16 +92,6 @@ Transcription affects the output a lot. Whisper does not always give perfect tra Even tiny amounts of pause between phrases in the sample audio can have a big impact. Try increasing or decreasing them or find a sample with different cadence. -### audio.cpp `model contract spec not found for family '...'` - -The audio.cpp server discovers `model_specs/.json` relative to its **process working directory**, so it must be started from the audio.cpp checkout. The hub starts it that way automatically, and the launch hint it prints is prefixed with `cd &&`. If you start `audiocpp_server` by hand, run it from the checkout root: - -```bash -cd app/audio.cpp && ./build/--release/bin/audiocpp_server --config server.json -``` - -If the error instead mentions a model path that does not exist, the model package was never downloaded — the hub's status table shows `installed (models missing)` for that case. Install it from the checkout (the exact command is in the convert-menu warning), e.g. `python tools/model_manager_v2.py install qwen3_tts_0_6b_base_q8_0`. - ## License MIT diff --git a/app/backends/audiocpp.py b/app/backends/audiocpp.py index feeb22a..d4c79b0 100755 --- a/app/backends/audiocpp.py +++ b/app/backends/audiocpp.py @@ -94,7 +94,7 @@ AUDIOCPP_DIR_NAME = "audio.cpp" AUDIOCPP_GIT_URL = "https://github.com/0xShug0/audio.cpp" # Sentinel returned by tui.confirm (via its cancel_value) when the user -# presses Esc on an overwrite prompt to go back to the checkout browser +# presses Esc on an overwrite prompt to go back to the wav-directory browser # instead of aborting the wizard. _GO_BACK = object() @@ -161,38 +161,6 @@ def _resolve_audiocpp_root(directory: Path) -> Optional[Path]: return None -def _audiocpp_root_status(directory: Path) -> Tuple[str, str]: - """TUI status describing the directory listed in the checkout browser.""" - if _resolve_audiocpp_root(directory) is not None: - return ("model_specs/ found here", "ok") - return ("No model_specs/ directory here", "warn") - - -def _audiocpp_root_preview(directory: Path) -> Optional[Tuple[str, str]]: - """TUI status for a highlighted subdirectory in the checkout browser.""" - if (directory / "model_specs").is_dir(): - return ("contains model_specs/", "ok") - return None - - -def _checkout_auto_select(entry: Path) -> Optional[Path]: - """Auto-accept a highlighted checkout in the TUI browser. - - A subdirectory named ``audio.cpp`` that already contains a - ``model_specs`` directory is the audio.cpp checkout root, so it is - accepted immediately on Enter/Right (as if ``[ Use this directory ]`` - had been pressed) instead of being descended into. Anything else - returns None so the user keeps browsing. This is only consulted - while auto-accepting is still enabled; after the user presses Esc to - go back, the browser is restarted inside the previously accepted - checkout and this callback is no longer passed, so a wrong guess can - be corrected. - """ - if entry.name == "audio.cpp" and (entry / "model_specs").is_dir(): - return entry - return None - - # Backend display order, with short descriptions. The backend name is padded # so the descriptions' dashes line up in the menu. _BACKEND_DESCRIPTIONS = ( @@ -1058,54 +1026,6 @@ def _wizard(stdscr, args: argparse.Namespace, parser: argparse.ArgumentParser "unused_entries": s["unused_entries"], } - def screen_no_checkout(): - """First screen when no checkout exists: clone or browse. - - No ``back_value``: Esc aborts the whole wizard (nothing before it). - """ - choice = tui.menu( - stdscr, "No audio.cpp checkout found", - [(f"Clone into ./app/{AUDIOCPP_DIR_NAME} " - f"(from {AUDIOCPP_GIT_URL})", "clone"), - ("Browse for an existing checkout", "browse")], - help_lines=[ - "audio.cpp hosts the TTS model families " - "this generator uses.", - "Clone it into the project's app " - "directory, or point at an existing " - "checkout."]) - if choice == "clone": - target = APP_DIR / AUDIOCPP_DIR_NAME - with tui.suspend(stdscr): - rc = common.git_clone(AUDIOCPP_GIT_URL, target) - if rc != 0: - raise _TuiError( - f"git clone failed (exit {rc}). Clone " - f"audio.cpp manually: git clone " - f"{AUDIOCPP_GIT_URL} {target}") - resolve_checkout(target) - else: - return screen_browse_checkout - return _after_families() - - def screen_browse_checkout(): - """Browse for an existing checkout (Esc returns to the clone menu).""" - audiocpp_dir = tui.browse_directory( - stdscr, "Select your audio.cpp directory", - validate=lambda p: None if _resolve_audiocpp_root(p) - else "No model_specs/ directory here", - info=_audiocpp_root_status, - preview=_audiocpp_root_preview, - help_lines=["The root folder of your audio.cpp checkout;", - "it is the one that contains model_specs/"], - start=Path.cwd(), - auto_select=_checkout_auto_select, - back_value=_GO_BACK) - if audiocpp_dir is _GO_BACK: - return tui.Wizard.BACK - resolve_checkout(audiocpp_dir) - return _after_families() - def screen_families(): """Pick TTS model families and packages (the modify tree).""" tree_families = _build_tree_families(s["catalog"]) @@ -1385,15 +1305,24 @@ def _wizard(stdscr, args: argparse.Namespace, parser: argparse.ArgumentParser return _finalize() # First screen: resolve the checkout directly when it already exists - # (the modify flow), so the wizard starts on a real screen. + # (the modify flow), so the wizard starts on a real screen. When no + # checkout exists, clone it into ./app/audio.cpp without asking, then + # continue the same way. audiocpp_dir = args.audiocpp_dir if audiocpp_dir is None: audiocpp_dir = find_local_checkout() if audiocpp_dir is None: - first = screen_no_checkout - else: - resolve_checkout(audiocpp_dir) - first = _after_families() + target = APP_DIR / AUDIOCPP_DIR_NAME + with tui.suspend(stdscr): + rc = common.git_clone(AUDIOCPP_GIT_URL, target) + if rc != 0: + raise _TuiError( + f"git clone failed (exit {rc}). Clone " + f"audio.cpp manually: git clone " + f"{AUDIOCPP_GIT_URL} {target}") + audiocpp_dir = target + resolve_checkout(audiocpp_dir) + first = _after_families() return tui.Wizard().run(first) diff --git a/app/backends/faster.py b/app/backends/faster.py index e209ff2..36121ff 100755 --- a/app/backends/faster.py +++ b/app/backends/faster.py @@ -225,30 +225,11 @@ def _wizard(stdscr, args: argparse.Namespace) -> Optional[dict]: wav_start = next(iter(ref_dirs)) s["wav_start"] = wav_start - def screen_install(): - choice = tui.confirm(stdscr, "faster-qwen3-tts is not installed. " - "pip install it now?", default=True, - cancel_value=_GO_BACK) - if choice is _GO_BACK: - return tui.Wizard.BACK - s["do_install"] = choice - return _after_install() - - def _after_install(): - if not _is_cloned() and not args.skip_clone: - return screen_clone - s["do_clone"] = False - return _after_clone() - - def screen_clone(): - choice = tui.confirm( - stdscr, f"faster-qwen3-tts repo not cloned. Clone it into " - f"./app/{FASTER_DIR_NAME}?", default=True, - cancel_value=_GO_BACK) - if choice is _GO_BACK: - return tui.Wizard.BACK - s["do_clone"] = choice - return _after_clone() + # Install and clone happen without asking: when the package or repo is + # missing (and not skipped by flag), the wizard just does it and moves + # to the next screen. + s["do_install"] = (not _is_installed()) and not args.skip_install + s["do_clone"] = (not _is_cloned()) and not args.skip_clone def _after_clone(): if args.input_dir is None: @@ -361,11 +342,7 @@ def _wizard(stdscr, args: argparse.Namespace) -> Optional[dict]: "plan": s["plan"], } - if not _is_installed() and not args.skip_install: - first = screen_install - else: - first = _after_install() - return tui.Wizard().run(first) + return tui.Wizard().run(_after_clone()) def _try_language(value: str) -> bool: diff --git a/app/backends/qwen.py b/app/backends/qwen.py index 3416ebe..f1e7c79 100644 --- a/app/backends/qwen.py +++ b/app/backends/qwen.py @@ -71,15 +71,6 @@ def _wizard(stdscr, args: argparse.Namespace) -> Optional[dict]: _GO_BACK = object() s: dict = {} - def screen_install(): - choice = tui.confirm(stdscr, "qwen-tts is not installed. " - "pip install it now?", default=True, - cancel_value=_GO_BACK) - if choice is _GO_BACK: - return tui.Wizard.BACK - s["do_install"] = choice - return _after_install() - def _after_install(): if args.port_custom is None: return screen_custom_port @@ -145,11 +136,11 @@ def _wizard(stdscr, args: argparse.Namespace) -> Optional[dict]: "speaker": s["speaker"], } - if not _is_installed() and not args.skip_install: - first = screen_install - else: - first = _after_install() - return tui.Wizard().run(first) + # pip install happens without asking: when the package is missing (and + # not skipped by flag), the wizard just does it and moves to the next + # screen. + s["do_install"] = (not _is_installed()) and not args.skip_install + return tui.Wizard().run(_after_install()) def _execute(settings: dict) -> int: diff --git a/app/tests/test_backends_audiocpp.py b/app/tests/test_backends_audiocpp.py index dd19cd5..b724c0d 100644 --- a/app/tests/test_backends_audiocpp.py +++ b/app/tests/test_backends_audiocpp.py @@ -370,40 +370,6 @@ class NormalizeDirArgTests(unittest.TestCase): self.assertEqual(result, Path("/tmp/foo").resolve()) -class CheckoutAutoSelectTests(unittest.TestCase): - """TUI browser auto-accept callback for an audio.cpp checkout.""" - - def setUp(self): - self._td = tempfile.TemporaryDirectory() - self.root = Path(self._td.name) - - def tearDown(self): - self._td.cleanup() - - def test_accepts_audio_cpp_containing_model_specs(self): - checkout = self.root / "audio.cpp" - checkout.mkdir() - (checkout / "model_specs").mkdir() - self.assertEqual(make_server._checkout_auto_select(checkout), - checkout) - - def test_rejects_audio_cpp_without_model_specs(self): - checkout = self.root / "audio.cpp" - checkout.mkdir() - self.assertIsNone(make_server._checkout_auto_select(checkout)) - - def test_rejects_other_name_even_with_model_specs(self): - other = self.root / "not-audiocpp" - other.mkdir() - (other / "model_specs").mkdir() - self.assertIsNone(make_server._checkout_auto_select(other)) - - def test_rejects_plain_directory(self): - plain = self.root / "somewhere" - plain.mkdir() - self.assertIsNone(make_server._checkout_auto_select(plain)) - - class DefaultModelIdTests(unittest.TestCase): def test_preferred_ids_for_tested_families(self): self.assertEqual(make_server.default_model_id("qwen3_tts"), "qwen") diff --git a/app/tests/test_tui.py b/app/tests/test_tui.py index b206de7..d5cbb7e 100644 --- a/app/tests/test_tui.py +++ b/app/tests/test_tui.py @@ -687,13 +687,6 @@ class FormTests(TuiTestCase): self.assertEqual(result, {"first": "a", "second": "bx"}) -def _accept_audio_cpp(entry: Path): - """auto_select callback that accepts an 'audio.cpp' checkout root.""" - if entry.name == "audio.cpp" and (entry / "model_specs").is_dir(): - return entry - return None - - class BrowseDirectoryTests(TuiTestCase): def setUp(self): super().setUp() @@ -704,16 +697,6 @@ class BrowseDirectoryTests(TuiTestCase): (self.root / name).mkdir() (self.root / "noise.txt").write_text("x", encoding="utf-8") - def _checkout_tree(self): - """A temp dir containing an 'audio.cpp' checkout + a sibling dir.""" - tmp = tempfile.TemporaryDirectory() - self.addCleanup(tmp.cleanup) - root = Path(tmp.name) - (root / "audio.cpp").mkdir() - (root / "audio.cpp" / "model_specs").mkdir() - (root / "other").mkdir() - return root - def test_listing_rows_left_justified(self): screen = FakeScreen(keys=[10]) chosen = tui.browse_directory(screen, "Pick", start=self.root) @@ -735,60 +718,6 @@ class BrowseDirectoryTests(TuiTestCase): self.assertEqual(chosen, (self.root / "alpha").resolve()) self.assert_inside_border(screen) - def test_enter_auto_accepts_matching_subdir(self): - root = self._checkout_tree() - # sel 0 = [ Use this directory ], 1 = .., 2 = audio.cpp/ - keys = [FakeCurses.KEY_DOWN, FakeCurses.KEY_DOWN, 10] - screen = FakeScreen(keys=keys) - chosen = tui.browse_directory(screen, "Pick", start=root, - auto_select=_accept_audio_cpp) - self.assertEqual(chosen, (root / "audio.cpp").resolve()) - - def test_right_auto_accepts_matching_subdir(self): - root = self._checkout_tree() - keys = [FakeCurses.KEY_DOWN, FakeCurses.KEY_DOWN, - FakeCurses.KEY_RIGHT] - screen = FakeScreen(keys=keys) - chosen = tui.browse_directory(screen, "Pick", start=root, - auto_select=_accept_audio_cpp) - self.assertEqual(chosen, (root / "audio.cpp").resolve()) - - def test_use_this_directory_ignores_auto_select(self): - # Enter on '[ Use this directory ]' must accept the listed dir - # without ever consulting auto_select. - root = self._checkout_tree() - calls = [] - - def callback(entry): - calls.append(entry) - return entry # would auto-accept any subdir if consulted - - screen = FakeScreen(keys=[10]) - chosen = tui.browse_directory(screen, "Pick", start=root, - auto_select=callback) - self.assertEqual(chosen, root.resolve()) - self.assertEqual(calls, []) - - def test_auto_select_returning_none_descends_normally(self): - # A non-matching subdir (or a None reply) keeps browsing: Enter - # descends into it, then '[ Use this directory ]' accepts it. - root = self._checkout_tree() - calls = [] - - def callback(entry): - calls.append(entry) - return None - - # sel 0 = use, 1 = .., 2 = audio.cpp/, 3 = other/ - keys = [FakeCurses.KEY_DOWN, FakeCurses.KEY_DOWN, - FakeCurses.KEY_DOWN, 10, 10] - screen = FakeScreen(keys=keys) - chosen = tui.browse_directory(screen, "Pick", start=root, - auto_select=callback) - self.assertEqual(chosen, (root / "other").resolve()) - # auto_select was consulted only for the highlighted 'other/' row. - self.assertEqual([p.name for p in calls], ["other"]) - def test_esc_returns_back_value(self): marker = object() screen = FakeScreen(keys=[27]) diff --git a/app/ui/tui.py b/app/ui/tui.py index 0a119e9..f1abea1 100644 --- a/app/ui/tui.py +++ b/app/ui/tui.py @@ -1002,8 +1002,6 @@ def browse_directory(scr, title: str, preview: Optional[Callable[[Path], Optional[Tuple[str, str]]]] = None, help_lines: Optional[Sequence[str]] = None, - auto_select: Optional[Callable[ - [Path], Optional[Path]]] = None, back_value: object = None ) -> Path: """Pick a directory DOS-browser style. @@ -1023,15 +1021,9 @@ def browse_directory(scr, title: str, listed directory's path — kind is "ok" (green), "warn" (yellow), "err" (red), "info" (dim) or "input". PREVIEW(directory) returns one for the highlighted subdirectory, shown on the status line. - AUTO_SELECT receives a highlighted subdirectory when the user - opens it (Enter, Right or 'l') and may return a Path to accept - immediately — as if '[ Use this directory ]' had been pressed on - it — instead of descending; returning None keeps browsing. This - lets a subdirectory that already looks like the target (e.g. an - 'audio.cpp' checkout containing 'model_specs/') be picked in one - keystroke. Esc (or 'q') aborts the wizard unless BACK_VALUE is given - (not None), in which case either key returns it so the caller can - fall back a screen. + Esc (or 'q') aborts the wizard unless BACK_VALUE is given (not + None), in which case either key returns it so the caller can fall + back a screen. """ footer = ("Up/Down = move Enter = open/use Left = parent " "e = type path Esc = cancel") @@ -1119,12 +1111,7 @@ def browse_directory(scr, title: str, highlight = current current = current.parent else: - entry = entries[sel - offset] - if auto_select is not None: - picked = auto_select(entry) - if picked is not None: - return picked - current = entry + current = entries[sel - offset] sel = 0 elif key in (curses.KEY_LEFT, ord("h"), ord("u"), curses.KEY_BACKSPACE, 8, 127): -- cgit v1.2.3