aboutsummaryrefslogtreecommitdiff
path: root/app
diff options
context:
space:
mode:
authorhistoria <historiavg@proton.me>2026-08-31 19:45:57 -0400
committerhistoria <historiavg@proton.me>2026-08-31 19:45:57 -0400
commit10e72d4960e865acf5346ab8cf518ed5844fe45c (patch)
treeadf8c10386b9da6280c247f1fed137ef1a514157 /app
parent4bd0282da65db9f118ef5250582ab67079fad538 (diff)
downloadtts-audiobook-generator-10e72d4960e865acf5346ab8cf518ed5844fe45c.tar.gz
feat: generate a book with all installed models to compare
Diffstat (limited to 'app')
-rw-r--r--app/converter/clients/audiocpp.py17
-rw-r--r--app/converter/converter.py33
-rw-r--r--app/tests/test_audiobook_cli.py233
-rw-r--r--app/tests/test_converter.py39
-rw-r--r--app/tests/test_hub.py288
-rw-r--r--app/tests/test_runview.py39
-rw-r--r--app/tests/test_tts.py80
-rw-r--r--app/ui/hub.py294
-rw-r--r--app/ui/runview.py28
9 files changed, 1005 insertions, 46 deletions
diff --git a/app/converter/clients/audiocpp.py b/app/converter/clients/audiocpp.py
index a1888bc..df1884d 100644
--- a/app/converter/clients/audiocpp.py
+++ b/app/converter/clients/audiocpp.py
@@ -406,7 +406,8 @@ class AudioCppTTSClient(BaseTTSClient):
model_id: Optional[str] = None,
instructions: Optional[str] = None,
request_options: Optional[Dict[str, str]] = None,
- quiet: bool = False):
+ quiet: bool = False,
+ unload_models: Optional[bool] = None):
super().__init__(chunks_dir, quiet=quiet)
self.api_url = (api_url or config.AUDIOCPP_API_URL).rstrip("/")
# Per-run model selection: the --model CLI flag (or the Generate
@@ -415,6 +416,12 @@ class AudioCppTTSClient(BaseTTSClient):
# don't require --model.
self.model_id = (model_id or "").strip()
self._model_id_explicit = bool(self.model_id)
+ # Unload previously-loaded server models at connect time: None
+ # follows the AUDIOCPP_UNLOAD_MODELS setting (read at connect, so
+ # a Settings change this session is honored); True/False force it
+ # regardless of the setting ("All (multiple generation)" runs pass
+ # True so each per-model conversion starts with a clean VRAM).
+ self._unload_models_override = unload_models
# Validate before connecting so bad values fail fast without a server.
self.language = normalize_language(
language if language is not None else config.LANGUAGE)
@@ -570,7 +577,10 @@ class AudioCppTTSClient(BaseTTSClient):
self._report(f"[INFO] Sending instruction with every request: {self.instructions}")
self._report("[INFO] Its effect (style, emotion, delivery) depends on the "
"model family; models without instruction support ignore it.")
- if config.AUDIOCPP_UNLOAD_MODELS:
+ unload = (config.AUDIOCPP_UNLOAD_MODELS
+ if self._unload_models_override is None
+ else self._unload_models_override)
+ if unload:
self._unload_server_models()
def _require_synthesis_task(self, models: List[Dict[str, str]]) -> None:
@@ -595,7 +605,8 @@ class AudioCppTTSClient(BaseTTSClient):
on its first request. Failures only warn: an older server without
the endpoint, or a busy one, must not block a working setup.
Controlled by config.AUDIOCPP_UNLOAD_MODELS (the TUI Settings
- "Unload models" option).
+ "Unload models" option), or forced per run via the unload_models
+ override ("All (multiple generation)" runs unload between models).
"""
request = urllib.request.Request(
f"{self.api_url}/v1/tasks/unload_all_models", data=b"",
diff --git a/app/converter/converter.py b/app/converter/converter.py
index bd5e477..a10a57d 100644
--- a/app/converter/converter.py
+++ b/app/converter/converter.py
@@ -212,6 +212,7 @@ class AudiobookConverter:
instructions: Optional[str] = None,
request_options: Optional[Dict[str, str]] = None,
api_url: Optional[str] = None,
+ unload_models: Optional[bool] = None,
progress: Optional[Callable[[dict], None]] = None,
cancel=None):
if speed <= 0:
@@ -243,6 +244,9 @@ class AudiobookConverter:
# them against the server-hosted model at connect time.
self.instructions = instructions
self.request_options = dict(request_options or {})
+ # audio.cpp only: force unloading previously-loaded server models
+ # at connect time (None follows the AUDIOCPP_UNLOAD_MODELS setting).
+ self.unload_models = unload_models
self._validate_configuration()
# Interactive reporting (the TUI run view): PROGRESS receives an
# event dict per state change and turns the clients' console prints
@@ -261,13 +265,15 @@ class AudiobookConverter:
# elsewhere. model_id picks the server entry per run
# (auto-selected on single-entry servers); instructions
# describe or style the voice, request_options pass
- # per-model controls through to the server.
+ # per-model controls through to the server. unload_models
+ # forces a pre-run model unload when not None ("All" runs).
self.tts = AudioCppTTSClient(chunks_dir=CHUNKS_FOLDER,
voice=voice, language=self.language,
model_id=model_id,
instructions=instructions,
request_options=self.request_options,
- api_url=api_url, quiet=quiet)
+ api_url=api_url, quiet=quiet,
+ unload_models=unload_models)
else:
# Qwen: the voice mode picks the request shape (built-in
# speaker, clone from a reference .wav, or a designed voice);
@@ -355,7 +361,6 @@ class AudiobookConverter:
voice_clone_ref_audio: Optional[str],
instructions: Optional[str] = None) -> str:
"""Narrator name used in output file names, without a server connection.
-
Custom voice mode uses the built-in speaker's display name; voice
clone mode uses the reference audio file's stem; the faster and
audiocpp backends use the server-side voice name (for audiocpp's
@@ -392,6 +397,19 @@ class AudiobookConverter:
return AudiobookConverter._sanitize_filename(
narrator, fallback="narrator").replace(" ", "_")
+ @staticmethod
+ def compute_model_tag(model_id: Optional[str]) -> str:
+ """Model id used in output file names, without a server connection.
+
+ "All (multiple generation)" runs name every output with the
+ generating model's id so the per-model files never collide
+ (e.g. ``dune_qwen3_tts_1_7b_base_q8_0_Vivian.m4b``). Pure (no I/O,
+ no server) so the pre-flight can compute the exact output names a
+ run would produce before spending time connecting to a TTS server.
+ """
+ return AudiobookConverter._sanitize_filename(
+ model_id or "", fallback="model").replace(" ", "_")
+
# ------------------------------------------------------------------
# Debug dumps (--debug)
# ------------------------------------------------------------------
@@ -804,6 +822,7 @@ class AudiobookConverter:
confirm: Optional[Callable[[str, bool], bool]] = None,
book_files: Optional[List[Path]] = None,
output_name: Optional[str] = None,
+ name_tag: Optional[str] = None,
) -> Tuple[List[Path], List[Tuple[Path, str]]]:
"""Discover books and ask every overwrite question up front.
@@ -822,7 +841,10 @@ class AudiobookConverter:
(a single --input-file book; still filtered to supported formats),
and OUTPUT_NAME overrides the computed output name with a verbatim
base name (--output-file's stem, no narrator tag or stem-collision
- suffix). Both default to the directory-scan behavior.
+ suffix). Both default to the directory-scan behavior. NAME_TAG, when
+ given, is inserted between the book stem and the narrator tag
+ ("All (multiple generation)" runs pass the sanitized model id, so
+ each model's outputs are named and planned separately).
"""
if book_files is None:
book_files = sorted(
@@ -854,7 +876,8 @@ class AudiobookConverter:
name = book_file.stem
if stem_counts[book_file.stem] > 1:
name = f"{book_file.stem}_{book_file.suffix.lstrip('.')}"
- names.append((book_file, f"{name}_{narrator_tag}"))
+ tag = f"{name_tag}_{narrator_tag}" if name_tag else narrator_tag
+ names.append((book_file, f"{name}_{tag}"))
# Ask every overwrite question up front, before any conversion
# starts, so the rest of the run is unattended.
diff --git a/app/tests/test_audiobook_cli.py b/app/tests/test_audiobook_cli.py
index a3f0b90..f5dddf4 100644
--- a/app/tests/test_audiobook_cli.py
+++ b/app/tests/test_audiobook_cli.py
@@ -16,6 +16,7 @@ import logging
import shutil
import sys
import tempfile
+import threading
import unittest
from pathlib import Path
from unittest.mock import MagicMock, patch
@@ -333,6 +334,238 @@ class ConvertWiringTests(unittest.TestCase):
self._convert(output_file=self.tmp / "dune.mp3")
+class AllModelsConvertTests(unittest.TestCase):
+ """convert(model_ids=...): the "All (multiple generation)" loop.
+
+ One AudiobookConverter per model, model-major, each forced to unload
+ previously-loaded server models (clean VRAM between models); a failed
+ book or a model that cannot start does not sink the remaining models;
+ the progress events are renumbered into one global book sequence
+ stamped with the generating model and one merged "done" is emitted.
+ """
+
+ def setUp(self):
+ self.tmp = Path(tempfile.mkdtemp(prefix="audiobook_all_"))
+ self.addCleanup(shutil.rmtree, self.tmp, True)
+ self.book = _make_book(self.tmp)
+ self._old_folders = (converter_mod.BOOKS_FOLDER,
+ converter_mod.AUDIOBOOKS_FOLDER)
+ self.addCleanup(self._restore_folders)
+
+ def _restore_folders(self):
+ converter_mod.BOOKS_FOLDER, converter_mod.AUDIOBOOKS_FOLDER = \
+ self._old_folders
+
+ def _convert(self, *, run_results=None, make_run=None, progress=None,
+ cancel=None, planned_by_model="default", instances=None,
+ **kwargs):
+ """Run convert() with AudiobookConverter mocked per model.
+
+ RUN_RESULTS gives each model's converter.run() return value in
+ construction order; MAKE_RUN, when given, builds each instance's
+ run() from the constructor kwargs (for event-emitting fakes).
+ INSTANCES, when given, is a list the per-model instances are
+ appended to. Returns (code, class mock, per-model ctor kwargs).
+ """
+ kwargs.setdefault("backend", "audiocpp")
+ kwargs.setdefault("model_ids", ["m1", "m2"])
+ kwargs.setdefault("model_voices", {"m1": "narrator", "m2": None})
+ if planned_by_model == "default":
+ kwargs.setdefault("planned_by_model", {
+ "m1": [(self.book, "book_m1_narrator")],
+ "m2": [(self.book, "book_m2_none")]})
+ else:
+ kwargs["planned_by_model"] = planned_by_model
+ kwargs.setdefault("book_files", [self.book])
+ if progress is not None:
+ kwargs["progress"] = progress
+ if cancel is not None:
+ kwargs["cancel"] = cancel
+ states = list(run_results or [])
+ ctor_kwargs = []
+ made = instances if instances is not None else []
+
+ def make_instance(*args, **ckwargs):
+ ckwargs = dict(ckwargs)
+ ctor_kwargs.append(ckwargs)
+ inst = MagicMock()
+ made.append(inst)
+ if make_run is not None:
+ inst.run.side_effect = make_run(ckwargs)
+ else:
+ result = states[len(made) - 1] \
+ if len(made) <= len(states) else True
+ inst.run.return_value = result
+ return inst
+
+ fake_class = MagicMock(side_effect=make_instance)
+ with patch.object(audiobook, "setup_logging"), \
+ patch.object(audiobook, "setup_directories"), \
+ patch.object(audiobook, "AudiobookConverter", fake_class):
+ code = audiobook.convert(**kwargs)
+ return code, fake_class, ctor_kwargs
+
+ def test_one_converter_per_model_with_its_own_voice(self):
+ # Model-major: every model runs its planned books before the next
+ # model starts, each with the voice the form adapted for it and a
+ # forced pre-run model unload (clean VRAM between models).
+ code, fake_class, ctors = self._convert()
+ self.assertEqual(code, 0)
+ self.assertEqual(fake_class.call_count, 2)
+ self.assertEqual(ctors[0]["model_id"], "m1")
+ self.assertEqual(ctors[0]["voice"], "narrator")
+ self.assertTrue(ctors[0]["unload_models"])
+ self.assertEqual(ctors[1]["model_id"], "m2")
+ self.assertIsNone(ctors[1]["voice"])
+ self.assertTrue(ctors[1]["unload_models"])
+
+ def test_planned_entries_reach_each_converter(self):
+ instances = []
+ self._convert(instances=instances)
+ self.assertEqual(instances[0]._planned,
+ [(self.book, "book_m1_narrator")])
+ self.assertEqual(instances[1]._planned,
+ [(self.book, "book_m2_none")])
+
+ def test_failed_model_does_not_sink_the_next(self):
+ # m1's conversion fails: the loop still constructs and runs m2,
+ # and the overall run reports failure (not every book succeeded).
+ code, fake_class, _ = self._convert(run_results=[False, True])
+ self.assertEqual(code, 1)
+ self.assertEqual(fake_class.call_count, 2)
+
+ def test_model_that_cannot_start_is_skipped(self):
+ # A connect-time failure (constructor raise) is reported and the
+ # remaining models still run.
+ started = []
+
+ def make_instance(*args, **ckwargs):
+ started.append(ckwargs["model_id"])
+ if ckwargs["model_id"] == "m1":
+ raise RuntimeError("voice 'x' is not available")
+ inst = MagicMock()
+ inst.run.return_value = True
+ return inst
+
+ fake_class = MagicMock(side_effect=make_instance)
+ with patch.object(audiobook, "setup_logging"), \
+ patch.object(audiobook, "setup_directories"), \
+ patch.object(audiobook, "AudiobookConverter", fake_class), \
+ patch.object(logging_kit, "log_traceback"):
+ code = audiobook.convert(
+ backend="audiocpp", model_ids=["m1", "m2"],
+ model_voices={"m1": "narrator", "m2": None},
+ planned_by_model={"m1": [(self.book, "book_m1_narrator")],
+ "m2": [(self.book, "book_m2_none")]},
+ book_files=[self.book])
+ self.assertEqual(started, ["m1", "m2"])
+ self.assertEqual(code, 1)
+
+ def test_events_are_renumbered_and_stamped_with_the_model(self):
+ # Book events carry one global index across all models; done/cancel
+ # events from the per-model converters are swallowed and one merged
+ # "done" is emitted at the end.
+ events = []
+
+ def make_run(ckwargs):
+ emit = ckwargs["progress"]
+
+ def run():
+ emit({"kind": "book", "index": 1, "total": 1,
+ "name": "book.txt"})
+ emit({"kind": "chunks", "total": 3})
+ emit({"kind": "chunk_done", "chunk": 1, "total": 3})
+ emit({"kind": "book_done", "name": "book.txt", "ok": True,
+ "files": [f"book_{ckwargs['model_id']}.mp3"]})
+ emit({"kind": "done", "ok": 1, "total": 1})
+ return True
+ return run
+
+ code, _, _ = self._convert(progress=events.append,
+ make_run=make_run)
+ self.assertEqual(code, 0)
+ kinds = [e["kind"] for e in events]
+ self.assertEqual(kinds, ["book", "chunks", "chunk_done", "book_done",
+ "book", "chunks", "chunk_done", "book_done",
+ "done"])
+ self.assertEqual(events[0],
+ {"kind": "book", "index": 1, "total": 2,
+ "name": "book.txt", "model": "m1"})
+ self.assertEqual(events[4],
+ {"kind": "book", "index": 2, "total": 2,
+ "name": "book.txt", "model": "m2"})
+ self.assertEqual(events[3]["model"], "m1")
+ self.assertEqual(events[7]["model"], "m2")
+ self.assertEqual(events[8],
+ {"kind": "done", "ok": 2, "total": 2,
+ "cancelled": False})
+
+ def test_cancellation_stops_the_remaining_models(self):
+ # m1's run sets the cancel event: the loop stops before m2 and the
+ # merged done reports the cancellation.
+ cancel = threading.Event()
+ events = []
+
+ def make_run(ckwargs):
+ emit = ckwargs["progress"]
+
+ def run():
+ emit({"kind": "book", "index": 1, "total": 1,
+ "name": "book.txt"})
+ emit({"kind": "book_done", "name": "book.txt", "ok": True,
+ "files": ["book_m1_narrator.mp3"]})
+ cancel.set()
+ return False
+ return run
+
+ code, fake_class, _ = self._convert(progress=events.append,
+ make_run=make_run, cancel=cancel)
+ self.assertEqual(fake_class.call_count, 1)
+ self.assertEqual(code, 1)
+ self.assertEqual(events[-1],
+ {"kind": "done", "ok": 1, "total": 2,
+ "cancelled": True})
+
+ def test_plans_are_computed_when_not_provided(self):
+ # Without planned_by_model (a scripted call) each model plans its
+ # own model-tagged outputs, with its own voice for the narrator
+ # tag the overwrite questions are asked about.
+ preflight = MagicMock(return_value=([self.book],
+ [(self.book, "dune")]))
+ fake_class = MagicMock()
+ fake_class.preflight_overwrites = preflight
+ # The converter class is mocked wholesale; the name-tag helper is
+ # a pure static method, so stand in the real behavior.
+ fake_class.compute_model_tag = staticmethod(
+ AudiobookConverter.compute_model_tag)
+
+ def make_instance(*args, **ckwargs):
+ inst = MagicMock()
+ inst.run.return_value = True
+ return inst
+ fake_class.side_effect = make_instance
+ with patch.object(audiobook, "setup_logging"), \
+ patch.object(audiobook, "setup_directories"), \
+ patch.object(audiobook, "AudiobookConverter", fake_class):
+ code = audiobook.convert(
+ backend="audiocpp", model_ids=["m1", "m2"],
+ model_voices={"m1": "narrator", "m2": None},
+ book_files=None)
+ self.assertEqual(code, 0)
+ self.assertEqual(preflight.call_count, 2)
+ first, second = preflight.call_args_list
+ self.assertEqual(first.kwargs["voice"], "narrator")
+ self.assertEqual(first.kwargs["name_tag"], "m1")
+ self.assertEqual(second.kwargs["voice"], None)
+ self.assertEqual(second.kwargs["name_tag"], "m2")
+
+ def test_nothing_planned_is_a_clean_noop(self):
+ code, fake_class, _ = self._convert(
+ planned_by_model={"m1": [], "m2": []})
+ self.assertEqual(code, 0)
+ fake_class.assert_not_called()
+
+
class ManagedServerWiringTests(unittest.TestCase):
"""convert(manage_server=True) boots and stops the server around the run.
diff --git a/app/tests/test_converter.py b/app/tests/test_converter.py
index eaf96ef..41b9cad 100644
--- a/app/tests/test_converter.py
+++ b/app/tests/test_converter.py
@@ -594,6 +594,45 @@ class PreflightOverwritesTests(unittest.TestCase):
self.assertEqual(len(book_files), 1)
self.assertEqual(planned, [])
+ def test_name_tag_inserts_the_model_into_output_names(self):
+ # "All (multiple generation)" runs plan each model separately: the
+ # sanitized model id sits between the book stem and the narrator
+ # tag, so the per-model outputs never collide.
+ with patch("builtins.input", side_effect=AssertionError("should not prompt")):
+ book_files, planned = AudiobookConverter.preflight_overwrites(
+ BACKEND_AUDIOCPP, "Vivian", VOICE_MODE_CUSTOM, None, "mp3",
+ name_tag="qwen3_tts_1_7b_base_q8_0")
+ self.assertEqual(planned,
+ [(book_files[0],
+ "book_qwen3_tts_1_7b_base_q8_0_Vivian")])
+
+ def test_name_tag_after_the_stem_collision_suffix(self):
+ (converter_mod.BOOKS_FOLDER / "book.epub").write_text("x",
+ encoding="utf-8")
+ with patch("builtins.input", side_effect=AssertionError("should not prompt")):
+ book_files, planned = AudiobookConverter.preflight_overwrites(
+ BACKEND_AUDIOCPP, "Vivian", VOICE_MODE_CUSTOM, None, "mp3",
+ name_tag="m1")
+ names = {name for _book, name in planned}
+ self.assertEqual(names, {"book_txt_m1_Vivian", "book_epub_m1_Vivian"})
+
+
+class ComputeModelTagTests(unittest.TestCase):
+ """compute_model_tag: the sanitized model id used in output names."""
+
+ def test_plain_id_passes_through(self):
+ self.assertEqual(AudiobookConverter.compute_model_tag(
+ "qwen3_tts_1_7b_base_q8_0"), "qwen3_tts_1_7b_base_q8_0")
+
+ def test_invalid_characters_and_spaces_become_underscores(self):
+ self.assertEqual(AudiobookConverter.compute_model_tag(
+ "model with spaces/colon"),
+ "model_with_spaces_colon")
+
+ def test_blank_falls_back(self):
+ self.assertEqual(AudiobookConverter.compute_model_tag(""), "model")
+ self.assertEqual(AudiobookConverter.compute_model_tag(None), "model")
+
class RunOverwritePromptTests(unittest.TestCase):
"""The full run() flow: prompts collected before any conversion starts."""
diff --git a/app/tests/test_hub.py b/app/tests/test_hub.py
index ef246ad..b1a91ed 100644
--- a/app/tests/test_hub.py
+++ b/app/tests/test_hub.py
@@ -10,7 +10,7 @@ import json
import tempfile
import unittest
from pathlib import Path
-from unittest.mock import patch
+from unittest.mock import MagicMock, patch
from backends import BackendInfo, BackendStatus, ServerSpec
from converter.clients import audiocpp as audiocpp_client
@@ -919,6 +919,18 @@ class ConvertFlowTests(unittest.TestCase):
remote_urls=remote_urls,
remote_models=list(remote_models or []))
+ def _mock_preflight(self, book="book.txt"):
+ """Replace the real books-folder scan with a canned plan.
+
+ Returns the mock so tests can assert how many per-model plans the
+ "All" flow computed (and with which name_tag/voice)."""
+ mk = MagicMock(return_value=([book], [(book, "planned")]))
+ patcher = patch.object(hub.AudiobookConverter, "preflight_overwrites",
+ mk)
+ patcher.start()
+ self.addCleanup(patcher.stop)
+ return mk
+
def test_audiocpp_remote_builds_one_form(self):
self._patch_remote(
[{"id": "higgs", "family": "higgs_audio_tts", "task": "tts"}],
@@ -996,8 +1008,12 @@ class ConvertFlowTests(unittest.TestCase):
self.assertEqual(choices[1],
("a-much-longer-model-id".ljust(22)
+ " clone", "a-much-longer-model-id"))
- self.assertEqual({label.index("clone") for label, _ in choices},
- {29})
+ # Two configured models: the "All (multiple generation)" pick
+ # closes the menu, plain-text without capability columns.
+ self.assertEqual(choices[2],
+ ("All (multiple generation)", hub.AUDIOCPP_MODEL_ALL))
+ self.assertEqual({label.index("clone") for label, _ in choices
+ if "clone" in label}, {29})
self.assertEqual({label.index("tts") for label, _ in choices
if "tts" in label}, {24})
# A plain qwen3_tts entry (no CustomVoice in the id) is clone-only.
@@ -1056,7 +1072,8 @@ class ConvertFlowTests(unittest.TestCase):
choices = self._field("model_id")["choices"]
base = "Qwen3-TTS-12Hz-1.7B-Base-GGUF"
design = "Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF"
- labels = {value: label for label, value in choices}
+ labels = {value: label for label, value in choices
+ if value != hub.AUDIOCPP_MODEL_ALL}
self.assertEqual(labels,
{base: base.ljust(len(design)) + " clone",
design: design.ljust(len(design))
@@ -1141,6 +1158,204 @@ class ConvertFlowTests(unittest.TestCase):
model_field["on_change"](fields)
self.assertEqual(voice_field["value"], "second")
+ # ------------------------------------------------------------------
+ # "All (multiple generation)" model pick
+ # ------------------------------------------------------------------
+
+ def test_model_menu_all_option_only_with_multiple_models(self):
+ # "All (multiple generation)" closes the model menu only when more
+ # than one model is configured: a single-model server has nothing
+ # to compare.
+ self._patch_remote([{"id": "solo", "family": "higgs_audio_tts",
+ "task": "tts"}], voices=["narrator"])
+ self._answer_form(backend="audiocpp-remote", model_id="solo",
+ audiocpp_voice="narrator", instructions="")
+ self._convert(None, [self._remote("audiocpp", "audio.cpp")])
+ self.assertEqual(self._field("model_id")["choices"],
+ [("solo tts clone", "solo")])
+ self._patch_remote(
+ [{"id": "alpha", "family": "higgs_audio_tts", "task": "tts"},
+ {"id": "beta", "family": "higgs_audio_tts", "task": "tts"}],
+ voices=["narrator"])
+ self._answer_form(backend="audiocpp-remote", model_id="alpha",
+ audiocpp_voice="narrator", instructions="")
+ self._convert(None, [self._remote("audiocpp", "audio.cpp")])
+ choices = self._field("model_id")["choices"]
+ self.assertEqual(choices[-1],
+ ("All (multiple generation)",
+ hub.AUDIOCPP_MODEL_ALL))
+ self.assertEqual(choices[0], ("alpha tts clone", "alpha"))
+
+ def test_all_pick_maps_one_run_per_model_with_the_picked_clone_voice(self):
+ # The "All" pick produces no single model/voice: the run receives
+ # the configured model list and each model's voice, with the
+ # picked server-side clone voice shared by every clone-capable
+ # model — and each model's overwrite plan computed with its
+ # model-tagged name.
+ self._patch_remote(
+ [{"id": "alpha", "family": "higgs_audio_tts", "task": "tts"},
+ {"id": "beta", "family": "chatterbox", "task": "tts"}],
+ voices=["narrator"])
+ mk_pre = self._mock_preflight()
+ self._answer_form(backend="audiocpp-remote",
+ model_id=hub.AUDIOCPP_MODEL_ALL,
+ audiocpp_voice="narrator", instructions="")
+ cmd = self._convert(None, [self._remote("audiocpp", "audio.cpp")])
+ kwargs = cmd[2]
+ self.assertEqual(kwargs["model_ids"], ["alpha", "beta"])
+ self.assertEqual(kwargs["model_voices"],
+ {"alpha": "narrator", "beta": "narrator"})
+ self.assertNotIn("model_id", kwargs)
+ self.assertNotIn("voice", kwargs)
+ self.assertEqual(kwargs["api_url"], "http://audiocpp.local:8080")
+ self.assertEqual(mk_pre.call_count, 2)
+ self.assertEqual(mk_pre.call_args_list[0].kwargs["name_tag"],
+ "alpha")
+ self.assertEqual(mk_pre.call_args_list[1].kwargs["name_tag"], "beta")
+ self.assertEqual(kwargs["book_files"], ["book.txt"])
+ self.assertEqual(kwargs["planned_by_model"],
+ {"alpha": [("book.txt", "planned")],
+ "beta": [("book.txt", "planned")]})
+ self.assertNotIn("planned", kwargs)
+
+ def test_all_voice_falls_back_per_capability(self):
+ # A CustomVoice entry cannot clone: it synthesizes with a built-in
+ # speaker (the pick when it names one, the first speaker when it
+ # names a clone voice); the Base entry cannot take a speaker name
+ # and clones with the picked server voice (the first when the pick
+ # names a speaker).
+ self._patch_remote(
+ [{"id": "Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF",
+ "family": "qwen3_tts", "task": "tts"},
+ {"id": "Qwen3-TTS-12Hz-1.7B-Base-GGUF", "family": "qwen3_tts",
+ "task": "tts"}],
+ voices=["narrator", "second"])
+ self._mock_preflight()
+ self._answer_form(
+ backend="audiocpp-remote", model_id=hub.AUDIOCPP_MODEL_ALL,
+ audiocpp_voice="narrator", instructions="")
+ cmd = self._convert(None, [self._remote("audiocpp", "audio.cpp")])
+ self.assertEqual(cmd[2]["model_voices"], {
+ "Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF":
+ hub.QWEN3_TTS_SPEAKERS[0],
+ "Qwen3-TTS-12Hz-1.7B-Base-GGUF": "narrator"})
+ self._answer_form(
+ backend="audiocpp-remote", model_id=hub.AUDIOCPP_MODEL_ALL,
+ audiocpp_voice="Vivian", instructions="")
+ cmd = self._convert(None, [self._remote("audiocpp", "audio.cpp")])
+ self.assertEqual(cmd[2]["model_voices"], {
+ "Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF": "Vivian",
+ "Qwen3-TTS-12Hz-1.7B-Base-GGUF": "narrator"})
+
+ def test_all_voice_union_offers_clone_voices_then_speakers(self):
+ # The Voice menu under "All" lists every model's clone voices
+ # first, then the built-in speakers while a speaker-capable model
+ # is configured; picking "All" keeps a voice the union offers.
+ self._patch_remote(
+ [{"id": "Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF",
+ "family": "qwen3_tts", "task": "tts"},
+ {"id": "alpha", "family": "chatterbox", "task": "tts"}],
+ voices=["narrator"])
+ self._mock_preflight()
+ self._answer_form(
+ backend="audiocpp-remote", model_id="alpha",
+ audiocpp_voice="narrator", instructions="")
+ self._convert(None, [self._remote("audiocpp", "audio.cpp")])
+ fields = self.tui.forms_seen[0][1]
+ voice_field = self._field("audiocpp_voice")
+ model_field = self._field("model_id")
+ self.assertTrue(voice_field["visible"](fields))
+ # Flip the Model pick to "All" (the form applies picks to the
+ # fields, firing on_change): the union lists every clone voice
+ # first, then the built-in speakers, and a clone pick survives.
+ voice_field["value"] = "narrator"
+ model_field["value"] = hub.AUDIOCPP_MODEL_ALL
+ model_field["on_change"](fields)
+ self.assertTrue(voice_field["visible"](fields))
+ self.assertEqual(voice_field["choices"](fields),
+ [(v, v) for v in
+ ["narrator"] + list(hub.QWEN3_TTS_SPEAKERS)])
+ self.assertEqual(voice_field["value"], "narrator")
+
+ def test_all_refuses_voice_design_model_without_instructions(self):
+ # A vdes entry in the "All" run needs the Instructions text its
+ # voice comes from: Generate! refuses with a pointed message
+ # instead of failing the run (or silently skipping the model).
+ self._patch_remote(
+ [{"id": "alpha", "family": "higgs_audio_tts", "task": "tts"},
+ {"id": "Qwen3-TTS-12Hz-1.7B-VoiceDesign-GGUF",
+ "family": "qwen3_tts", "task": "vdes"}],
+ voices=["narrator"])
+ self._mock_preflight()
+ self._answer_form(
+ backend="audiocpp-remote", model_id=hub.AUDIOCPP_MODEL_ALL,
+ audiocpp_voice="narrator", instructions="")
+ self._convert(None, [self._remote("audiocpp", "audio.cpp")])
+ fields = self.tui.forms_seen[0][1]
+ voice_field = self._field("audiocpp_voice")
+ instructions_field = self._field("instructions")
+ # The form applies the "All" pick to the field before validating.
+ self._field("model_id")["value"] = hub.AUDIOCPP_MODEL_ALL
+ error = voice_field["validate"]("narrator")
+ self.assertIsNotNone(error)
+ self.assertIn("voice design", error)
+ self.assertIn("Instructions", error)
+ self.assertEqual(instructions_field["validate"](""), error)
+ self.assertIsNone(instructions_field["validate"]("A warm narrator"))
+
+ def test_all_refuses_clone_only_model_without_voices(self):
+ # A clone-only family (chatterbox) whose server lists no voices
+ # cannot run in the "All" set: Generate! refuses naming the model
+ # (with a description the run would go — instruction voice).
+ self._patch_remote(
+ [{"id": "alpha", "family": "higgs_audio_tts", "task": "tts"},
+ {"id": "beta", "family": "chatterbox", "task": "tts"}],
+ voices=[])
+ self._mock_preflight()
+ self._answer_form(
+ backend="audiocpp-remote", model_id=hub.AUDIOCPP_MODEL_ALL,
+ audiocpp_voice="", instructions="")
+ self._convert(None, [self._remote("audiocpp", "audio.cpp")])
+ fields = self.tui.forms_seen[0][1]
+ voice_field = self._field("audiocpp_voice")
+ # The form applies the "All" pick to the field before validating.
+ self._field("model_id")["value"] = hub.AUDIOCPP_MODEL_ALL
+ error = voice_field["validate"]("")
+ self.assertIsNotNone(error)
+ self.assertIn("beta", error)
+ self.assertIn("Instructions", error)
+ # With a description the clone-only model designs its voice from
+ # it (instruction-voice mode): the run is accepted.
+ self._field("instructions")["value"] = "A warm narrator"
+ self.assertIsNone(voice_field["validate"](""))
+
+ def test_all_pick_works_on_the_managed_entry(self):
+ # The managed entry plans the same "All" run from server.json:
+ # voice_dir stems feed the union and the picked clone voice is
+ # shared by both clone-capable models.
+ with tempfile.TemporaryDirectory() as td:
+ root = Path(td)
+ (root / "server.json").write_text(json.dumps({
+ "models": [{"id": "qwen-1_7b", "family": "qwen3_tts",
+ "task": "tts"},
+ {"id": "qwen-0_6b", "family": "qwen3_tts",
+ "task": "tts"}],
+ "voice_dir": str(root),
+ }), encoding="utf-8")
+ (root / "Narrator.wav").write_bytes(b"x")
+ with patch.object(hub.audiocpp_backend, "find_local_checkout",
+ return_value=root):
+ self._mock_preflight()
+ self._answer_form(backend="audiocpp",
+ model_id=hub.AUDIOCPP_MODEL_ALL,
+ audiocpp_voice="Narrator", instructions="")
+ cmd = self._convert(None,
+ [self._ready("audiocpp", "audio.cpp")])
+ kwargs = cmd[2]
+ self.assertEqual(kwargs["model_ids"], ["qwen-1_7b", "qwen-0_6b"])
+ self.assertEqual(kwargs["model_voices"],
+ {"qwen-1_7b": "Narrator", "qwen-0_6b": "Narrator"})
+
def test_audiocpp_local_model_switch_keeps_the_picked_voice(self):
# The managed entry's voice list is shared by every model in
# server.json, so switching models keeps the picked voice.
@@ -2359,6 +2574,71 @@ class PreflightTests(unittest.TestCase):
with self.assertRaises(hub._BackToForm):
confirm("overwrite?", True)
+ # -- "All (multiple generation)": one plan per model ----------------
+
+ def _all_cmd(self):
+ return ("convert", "audiocpp", {
+ "model_ids": ["m1", "m2"],
+ "model_voices": {"m1": "narrator", "m2": None},
+ "output_format": "mp3", "clone": None})
+
+ def test_all_run_plans_each_model_and_stashes_planned_by_model(self):
+ stdscr = object()
+ cmd = self._all_cmd()
+ with patch.object(hub.AudiobookConverter, "preflight_overwrites",
+ side_effect=[(["book.txt"],
+ [("book.txt", "book_m1_narrator")]),
+ (["book.txt"],
+ [("book.txt", "book_m2_designed")])]) \
+ as mk_pre:
+ self.assertTrue(hub._preflight(stdscr, cmd))
+ self.assertEqual(mk_pre.call_count, 2)
+ first, second = mk_pre.call_args_list
+ # Each model plans with its own voice (so its narrator tag — and
+ # therefore its overwrite questions — match the real run) and its
+ # model-tagged output name.
+ self.assertEqual(first.kwargs["voice"], "narrator")
+ self.assertEqual(first.kwargs["voice_mode"],
+ hub.voice_mode_for("audiocpp", "narrator",
+ None, None))
+ self.assertEqual(first.kwargs["name_tag"], "m1")
+ self.assertIsNone(second.kwargs["voice"])
+ self.assertEqual(second.kwargs["voice_mode"],
+ hub.voice_mode_for("audiocpp", None, None, None))
+ self.assertEqual(second.kwargs["name_tag"], "m2")
+ self.assertEqual(cmd[2]["book_files"], ["book.txt"])
+ self.assertEqual(cmd[2]["planned_by_model"],
+ {"m1": [("book.txt", "book_m1_narrator")],
+ "m2": [("book.txt", "book_m2_designed")]})
+ self.assertNotIn("planned", cmd[2])
+
+ def test_all_run_no_books_flashes_and_returns_false(self):
+ stdscr = object()
+ with patch.object(hub.AudiobookConverter, "preflight_overwrites",
+ return_value=([], [])), \
+ patch.object(hub.tui, "flash") as mk_flash:
+ self.assertFalse(hub._preflight(stdscr, self._all_cmd()))
+ mk_flash.assert_called_once()
+
+ def test_all_run_all_skipped_flashes_and_returns_false(self):
+ stdscr = object()
+ with patch.object(hub.AudiobookConverter, "preflight_overwrites",
+ return_value=(["book.txt"], [])), \
+ patch.object(hub.tui, "flash") as mk_flash:
+ self.assertFalse(hub._preflight(stdscr, self._all_cmd()))
+ mk_flash.assert_called_once()
+
+ def test_all_run_confirm_esc_raises_back_to_form(self):
+ stdscr = object()
+ with patch.object(hub.AudiobookConverter, "preflight_overwrites",
+ return_value=(["book.txt"],
+ [("book.txt", "x")])) as mk_pre:
+ hub._preflight(stdscr, self._all_cmd())
+ confirm = mk_pre.call_args.kwargs["confirm"]
+ with patch.object(hub.tui, "confirm", return_value=hub._CANCEL):
+ with self.assertRaises(hub._BackToForm):
+ confirm("overwrite?", True)
+
class DispatchConversionTests(unittest.TestCase):
"""_Hub._run_conversion: builds the config and runs the run view."""
diff --git a/app/tests/test_runview.py b/app/tests/test_runview.py
index 5c0dc38..f8ab452 100644
--- a/app/tests/test_runview.py
+++ b/app/tests/test_runview.py
@@ -140,6 +140,31 @@ class StateTransitionTests(_FakeTui, unittest.TestCase):
view.handle_event({"kind": "server_stopped"})
self.assertEqual(view.server, "stopped")
+ def test_all_run_book_events_carry_the_model(self):
+ # "All (multiple generation)" runs stamp the generating model onto
+ # the book and book_done/book_failed events; the view keeps it for
+ # the progress line and the summary rows.
+ view, _ = self.make_view()
+ view.handle_event({"kind": "book", "index": 2, "total": 4,
+ "name": "book.txt", "model": "m1"})
+ self.assertEqual(view.book, (2, 4, "book.txt", "m1"))
+ view.handle_event({"kind": "book_done", "name": "book.txt",
+ "ok": True, "files": ["book_m1_Vivian.mp3"],
+ "model": "m1"})
+ self.assertEqual(view.book_results,
+ [("book.txt", True, ["book_m1_Vivian.mp3"], "",
+ "m1")])
+ view.handle_event({"kind": "book", "index": 3, "total": 4,
+ "name": "book.txt", "model": "m2"})
+ view.handle_event({"kind": "book_failed", "name": "book.txt",
+ "error": "chunk failed", "model": "m2"})
+ self.assertEqual(view.book_results[-1],
+ ("book.txt", False, [], "chunk failed", "m2"))
+ # A plain (single-model) run carries no model: the field stays None.
+ view.handle_event({"kind": "book", "index": 4, "total": 4,
+ "name": "book.txt"})
+ self.assertEqual(view.book, (4, 4, "book.txt", None))
+
class LogAppenderTests(_FakeTui, unittest.TestCase):
"""_LogAppender: stray console output survives in the run's log file."""
@@ -234,6 +259,20 @@ class RenderTests(_FakeTui, unittest.TestCase):
self.assertIn("book.txt", text)
self.assertIn("press any key", text)
+ def test_summary_screen_names_the_generating_model(self):
+ # An "All" run's summary rows distinguish which model generated
+ # each result (the progress line shows the model too).
+ view, screen = self.make_view()
+ view.handle_event({"kind": "book", "index": 1, "total": 2,
+ "name": "book.txt", "model": "m1"})
+ view.handle_event({"kind": "book_done", "name": "book.txt",
+ "ok": True, "files": ["book_m1_Vivian.mp3"],
+ "model": "m1"})
+ view.handle_event({"kind": "done", "ok": 1, "total": 2})
+ view.render()
+ text = self._strings(screen)
+ self.assertIn("book.txt — m1", text)
+
def test_stopping_status_shows_elapsed(self):
view, screen = self.make_view()
view.handle_event({"kind": "book", "index": 1, "total": 1,
diff --git a/app/tests/test_tts.py b/app/tests/test_tts.py
index 67c81f8..3c039ab 100644
--- a/app/tests/test_tts.py
+++ b/app/tests/test_tts.py
@@ -1957,6 +1957,7 @@ class AudioCppUnloadModelsTests(unittest.TestCase):
client.instruction_voice = False
client.speaker_mode = False
client.instructions = ""
+ client._unload_models_override = None
with patch.object(client, "_check_health"), \
patch.object(client, "_list_models",
return_value=[{"id": client.model_id,
@@ -1971,6 +1972,70 @@ class AudioCppUnloadModelsTests(unittest.TestCase):
client._connect()
mock_unload.assert_called_once()
+ def test_connect_unload_override_forces_unload(self):
+ # "All (multiple generation)" runs force the unload regardless of
+ # the AUDIOCPP_UNLOAD_MODELS setting, so each model starts clean.
+ client = AudioCppTTSClient.__new__(AudioCppTTSClient)
+ client.api_url = "http://127.0.0.1:8080"
+ client.model_id = _AUDIOCPP_MODEL_ID
+ client.preset_mode = True
+ client.voice = "narrator"
+ client.language = "English"
+ client._seed = -1
+ client.family = "qwen3_tts"
+ client.task = AUDIOCPP_TASK_TTS
+ client.profile = AUDIOCPP_FAMILY_PROFILES["qwen3_tts"]
+ client.design_mode = False
+ client.instruction_voice = False
+ client.speaker_mode = False
+ client.instructions = ""
+ client._unload_models_override = True
+ with patch.object(client, "_check_health"), \
+ patch.object(client, "_list_models",
+ return_value=[{"id": client.model_id,
+ "family": "qwen3_tts",
+ "task": "tts"}]), \
+ patch.object(client, "_auto_pick_model_id"), \
+ patch.object(client, "_require_model_id"), \
+ patch.object(client, "_resolve_family"), \
+ patch.object(client, "_resolve_task"), \
+ patch.object(client, "_check_voice"), \
+ patch.object(config, "AUDIOCPP_UNLOAD_MODELS", False), \
+ patch.object(client, "_unload_server_models") as mock_unload:
+ client._connect()
+ mock_unload.assert_called_once()
+
+ def test_connect_unload_override_false_skips_unload(self):
+ client = AudioCppTTSClient.__new__(AudioCppTTSClient)
+ client.api_url = "http://127.0.0.1:8080"
+ client.model_id = _AUDIOCPP_MODEL_ID
+ client.preset_mode = True
+ client.voice = "narrator"
+ client.language = "English"
+ client._seed = -1
+ client.family = "qwen3_tts"
+ client.task = AUDIOCPP_TASK_TTS
+ client.profile = AUDIOCPP_FAMILY_PROFILES["qwen3_tts"]
+ client.design_mode = False
+ client.instruction_voice = False
+ client.speaker_mode = False
+ client.instructions = ""
+ client._unload_models_override = False
+ with patch.object(client, "_check_health"), \
+ patch.object(client, "_list_models",
+ return_value=[{"id": client.model_id,
+ "family": "qwen3_tts",
+ "task": "tts"}]), \
+ patch.object(client, "_auto_pick_model_id"), \
+ patch.object(client, "_require_model_id"), \
+ patch.object(client, "_resolve_family"), \
+ patch.object(client, "_resolve_task"), \
+ patch.object(client, "_check_voice"), \
+ patch.object(config, "AUDIOCPP_UNLOAD_MODELS", True), \
+ patch.object(client, "_unload_server_models") as mock_unload:
+ client._connect()
+ mock_unload.assert_not_called()
+
def test_connect_skips_unload_when_disabled(self):
client = AudioCppTTSClient.__new__(AudioCppTTSClient)
client.api_url = "http://127.0.0.1:8080"
@@ -1986,6 +2051,7 @@ class AudioCppUnloadModelsTests(unittest.TestCase):
client.instruction_voice = False
client.speaker_mode = False
client.instructions = ""
+ client._unload_models_override = None
with patch.object(client, "_check_health"), \
patch.object(client, "_list_models",
return_value=[{"id": client.model_id,
@@ -2029,7 +2095,8 @@ class BackendWiringTests(unittest.TestCase):
model_id=None,
instructions=None,
request_options={},
- api_url=None, quiet=False)
+ api_url=None, quiet=False,
+ unload_models=None)
mock_faster.assert_not_called()
mock_qwen.assert_not_called()
@@ -2042,7 +2109,8 @@ class BackendWiringTests(unittest.TestCase):
model_id=None,
instructions=None,
request_options={},
- api_url=None, quiet=False)
+ api_url=None, quiet=False,
+ unload_models=None)
def test_audiocpp_backend_model_id_is_wired_through(self):
with patch("converter.converter.AudioCppTTSClient") as mock_audiocpp:
@@ -2053,7 +2121,8 @@ class BackendWiringTests(unittest.TestCase):
chunks_dir=converter_mod.CHUNKS_FOLDER,
voice="narrator", language=config.LANGUAGE,
model_id="higgs", instructions=None,
- request_options={}, api_url=None, quiet=False)
+ request_options={}, api_url=None, quiet=False,
+ unload_models=None)
def test_audiocpp_backend_instructions_and_options_are_wired_through(self):
with patch("converter.converter.AudioCppTTSClient") as mock_audiocpp:
@@ -2068,7 +2137,7 @@ class BackendWiringTests(unittest.TestCase):
model_id=None,
instructions="A warm adult narrator",
request_options={"emotion": "neutral", "speed": "1.1"},
- api_url=None, quiet=False)
+ api_url=None, quiet=False, unload_models=None)
def test_qwen_backend_uses_qwen_client(self):
with patch("converter.converter.FasterTTSClient") as mock_faster, \
@@ -2119,7 +2188,8 @@ class BackendWiringTests(unittest.TestCase):
chunks_dir=converter_mod.CHUNKS_FOLDER,
voice="narrator", language=config.LANGUAGE, model_id=None,
instructions=None, request_options={},
- api_url="http://10.0.0.5:8080", quiet=False)
+ api_url="http://10.0.0.5:8080", quiet=False,
+ unload_models=None)
with patch("converter.converter.FasterTTSClient") as mock_faster:
AudiobookConverter(voice_mode=VOICE_MODE_CLONE,
backend=BACKEND_FASTER, voice="narrator",
diff --git a/app/ui/hub.py b/app/ui/hub.py
index de5701f..9bd136a 100644
--- a/app/ui/hub.py
+++ b/app/ui/hub.py
@@ -74,6 +74,11 @@ from ui import runview, taskview, tui
_CANCEL = object() # sentinel: a convert preflight confirm backed out
+# The Generate form's audio.cpp Model pick for "All (multiple generation)":
+# one conversion per configured model (model-major), with model-tagged
+# output names. A sentinel string, distinct from every real model id.
+AUDIOCPP_MODEL_ALL = "__all__"
+
class _BackToForm(Exception):
"""Raised when Esc backs out of a preflight confirm (re-show the form)."""
@@ -941,8 +946,13 @@ def _preflight(stdscr, cmd: tuple) -> bool:
kwargs (``book_files``/``planned``) for ``audiobook.convert``. Returns
False when nothing would be converted (a flash explains why), so the
user stays in the menu instead of entering an empty run.
+
+ An "All (multiple generation)" run (``model_ids`` in the kwargs) is
+ planned by _preflight_all instead: one plan per model.
"""
_kind, backend, kwargs = cmd
+ if kwargs.get("model_ids"):
+ return _preflight_all(stdscr, backend, kwargs)
voice_mode = voice_mode_for(backend, kwargs.get("voice"),
kwargs.get("clone"),
kwargs.get("instructions"))
@@ -975,6 +985,59 @@ def _preflight(stdscr, cmd: tuple) -> bool:
return True
+def _preflight_all(stdscr, backend: str, kwargs: dict) -> bool:
+ """Run the overwrite checks for an "All (multiple generation)" run.
+
+ Plans one conversion per model: every model's output names carry its
+ model tag and its own adapted voice (the mapper's ``model_voices``),
+ so each model's overwrites are asked — and accepted — separately, all
+ up front (a cancel returns to the form). Records the union book list
+ as ``book_files`` and the per-model plans as ``planned_by_model`` on
+ the command kwargs for ``audiobook.convert``. Returns False when
+ nothing would be converted (a flash explains why), so the user stays
+ in the menu instead of entering an empty run.
+ """
+ model_ids = kwargs.get("model_ids") or []
+ model_voices = kwargs.get("model_voices") or {}
+ instructions = kwargs.get("instructions")
+
+ def confirm(message: str, default: bool) -> bool:
+ answer = tui.confirm(stdscr, message, default=default,
+ cancel_value=_CANCEL)
+ if answer is _CANCEL:
+ raise _BackToForm()
+ return answer
+
+ book_files: list = []
+ planned_by_model: dict = {}
+ with contextlib.redirect_stdout(io.StringIO()):
+ for model_id in model_ids:
+ voice = model_voices.get(model_id)
+ voice_mode = voice_mode_for(backend, voice,
+ kwargs.get("clone"), instructions)
+ books, planned = AudiobookConverter.preflight_overwrites(
+ backend=backend, voice=voice, voice_mode=voice_mode,
+ voice_clone_ref_audio=kwargs.get("clone"),
+ output_format=kwargs.get("output_format")
+ or config.AUDIO_FORMAT,
+ instructions=instructions, confirm=confirm,
+ name_tag=AudiobookConverter.compute_model_tag(model_id))
+ if not book_files:
+ book_files = books
+ planned_by_model[model_id] = planned
+ if not book_files:
+ tui.flash(stdscr, "No books to convert. Add a .txt, .pdf or .epub "
+ "file to the input folder first.")
+ return False
+ if not any(planned_by_model.values()):
+ tui.flash(stdscr, "Nothing to convert — every existing output was "
+ "kept.")
+ return False
+ kwargs["book_files"] = book_files
+ kwargs["planned_by_model"] = planned_by_model
+ return True
+
+
def _gate_backend(field: dict, key: str) -> Callable:
"""A visible() that shows FIELD only when the Backend field is KEY.
@@ -1055,6 +1118,18 @@ def _audiocpp_fields(stdscr, api_url: Optional[str] = None,
the field keys ("" for the managed entry) so two entries of this
backend can share one form without overwriting each other.
+ With more than one model configured, the Model menu closes with
+ "All (multiple generation)" (AUDIOCPP_MODEL_ALL): the run then
+ generates every book once per model, with model-tagged output names.
+ The single Voice pick is sent to every model that accepts it (the
+ same server-side clone voice, or a built-in speaker name on
+ CustomVoice entries); models the pick cannot serve fall back to
+ their own default (first speaker / first server voice / no voice —
+ design entries take the Instructions text), and Generate! refuses
+ the combinations that cannot work (a design model without
+ Instructions; a clone-only model without server voices or an
+ instruction-defined voice). See the "All" helpers below.
+
The Voice field tracks the selected entry's capability — built-in
speakers on CustomVoice, the server's clone voices on every other
entry. A clone-capable entry whose server lists no voices cannot be
@@ -1168,20 +1243,136 @@ def _audiocpp_fields(stdscr, api_url: Optional[str] = None,
return next((m for m in models if m.get("id") == model_id),
models[0])
- def model_capability(fields) -> str:
- entry = model_entry(fields)
+ def entry_capability(entry: dict) -> str:
+ """How ENTRY's voice is supplied (speaker/clone/design)."""
return audiocpp_entry_voice_capability(
entry.get("family") or "", entry.get("task") or "tts",
entry.get("id") or "")
+ def all_selected(fields) -> bool:
+ """True when the Model pick is "All (multiple generation)"."""
+ return _field_value(fields, prefix + "model_id") == AUDIOCPP_MODEL_ALL
+
+ def model_capability(fields) -> str:
+ return entry_capability(model_entry(fields))
+
def model_voice_policy(fields) -> str:
"""The selected entry's family voice policy (required/optional/none)."""
return audiocpp_family_voice_policy(
model_entry(fields).get("family") or "")
+ # -- "All (multiple generation)" ------------------------------------
+ # One conversion per configured model: the single Voice pick is used
+ # by every model that accepts it — the same server-side clone voice
+ # flows into every clone-capable family, a built-in speaker name into
+ # every CustomVoice entry — and each remaining model falls back to
+ # its own sensible default (first built-in speaker, first server
+ # voice, or no voice at all: design entries and pure-TTS families
+ # take no voice, the client then designs from Instructions or
+ # synthesizes plainly).
+
+ def any_voice_model() -> bool:
+ """True when at least one configured model takes a voice pick."""
+ return any(
+ entry_capability(m) == AUDIOCPP_VOICE_SPEAKER
+ or (entry_capability(m) == AUDIOCPP_VOICE_CLONE
+ and audiocpp_family_voice_policy(
+ m.get("family") or "") != AUDIOCPP_VOICE_NONE)
+ for m in models)
+
+ def any_design_model() -> bool:
+ """True when at least one configured model designs its voice."""
+ return any(entry_capability(m) == AUDIOCPP_VOICE_DESIGN
+ for m in models)
+
+ def all_voice_union() -> list:
+ """Every voice an "All" run can offer.
+
+ Each clone-capable family's server voices first (shared by all of
+ them on the managed entry; per-model on a remote one), then the
+ built-in speakers when a speaker-capable model is configured.
+ Duplicates removed, order stable — a clone voice leads the list,
+ matching the pick-falls-back rules.
+ """
+ union = []
+ for m in models:
+ if entry_capability(m) != AUDIOCPP_VOICE_CLONE \
+ or audiocpp_family_voice_policy(
+ m.get("family") or "") == AUDIOCPP_VOICE_NONE:
+ continue
+ for voice in voices_for(m.get("id")):
+ if voice not in union:
+ union.append(voice)
+ if any(entry_capability(m) == AUDIOCPP_VOICE_SPEAKER
+ for m in models):
+ for speaker in QWEN3_TTS_SPEAKERS:
+ if speaker not in union:
+ union.append(speaker)
+ return union
+
+ def all_voice_for(model_id: str, picked: Optional[str]) -> Optional[str]:
+ """The voice to send for MODEL_ID in an "All" run.
+
+ The picked voice wins wherever the model accepts it; models the
+ pick cannot serve fall back to their own default: the first
+ built-in speaker (CustomVoice) or first server voice (cloning),
+ or no voice at all (design entries and voice-less clone families
+ — the client then designs the voice from Instructions or
+ synthesizes plainly).
+ """
+ entry = next((m for m in models if m.get("id") == model_id),
+ models[0])
+ capability = entry_capability(entry)
+ if capability == AUDIOCPP_VOICE_DESIGN:
+ return None
+ if capability == AUDIOCPP_VOICE_SPEAKER:
+ if picked and picked in QWEN3_TTS_SPEAKERS:
+ return picked
+ return QWEN3_TTS_SPEAKERS[0]
+ # Clone capability; the family policy decides whether a voice
+ # exists at all.
+ if audiocpp_family_voice_policy(
+ entry.get("family") or "") == AUDIOCPP_VOICE_NONE:
+ return None
+ voices = voices_for(model_id)
+ if picked and picked in voices:
+ return picked
+ return voices[0] if voices else None
+
+ def all_voice_problem() -> Optional[str]:
+ """Why an "All" run cannot start with the current settings, or None.
+
+ Refuses when a voice design model is configured without the
+ Instructions text its voice comes from, and when a clone-only
+ model has neither server voices nor an instruction-defined voice.
+ Every other mismatch is resolved by all_voice_for's per-model
+ fallback instead.
+ """
+ instructions_value = str(_field_value(fields, prefix + "instructions")
+ or "").strip()
+ if any_design_model() and not instructions_value:
+ return ("The 'All' run includes a voice design model — "
+ "describe the voice in Instructions")
+ for m in models:
+ if entry_capability(m) != AUDIOCPP_VOICE_CLONE:
+ continue
+ if audiocpp_family_voice_policy(
+ m.get("family") or "") != AUDIOCPP_VOICE_REQUIRED:
+ continue
+ if voices_for(m.get("id")) or instructions_value:
+ continue
+ return (f"No voices are available to clone for "
+ f"'{m.get('id')}' — configure voices on the server, "
+ "describe one in Instructions, or pick a single model")
+ return None
+
def reset_voice(fields) -> None:
"""Re-point the Voice field at the newly selected model's voice.
+ With "All" picked, the pick survives when the union of every
+ model's voices still offers it; otherwise it falls back to the
+ union's first entry (a server clone voice when one exists).
+
A model switch that keeps the same voice list (two clone entries
sharing one server's voices) keeps the current pick: only a value
the new list cannot offer is re-pointed at its default. Families
@@ -1190,6 +1381,12 @@ def _audiocpp_fields(stdscr, api_url: Optional[str] = None,
"""
voice_field = next(f for f in fields
if f.get("key") == prefix + "audiocpp_voice")
+ if all_selected(fields):
+ voices = all_voice_union()
+ if voice_field.get("value") in voices:
+ return
+ voice_field["value"] = voices[0] if voices else ""
+ return
capability = model_capability(fields)
if capability == AUDIOCPP_VOICE_DESIGN \
or model_voice_policy(fields) == AUDIOCPP_VOICE_NONE:
@@ -1213,6 +1410,8 @@ def _audiocpp_fields(stdscr, api_url: Optional[str] = None,
voice_field["value"] = voices[0] if voices else ""
def voice_choices(fields) -> list:
+ if all_selected(fields):
+ return [(v, v) for v in all_voice_union()]
capability = model_capability(fields)
if capability == AUDIOCPP_VOICE_SPEAKER:
# Built-in Qwen3-TTS CustomVoice speakers; no server query needed.
@@ -1246,12 +1445,19 @@ def _audiocpp_fields(stdscr, api_url: Optional[str] = None,
def voice_validate(value):
"""Refuse Generate! when this entry's clone voice is unavailable.
+ With "All" picked the same check runs across every configured
+ model (design models need Instructions; clone-only models need
+ server voices or an instruction-defined voice) — see
+ all_voice_problem.
+
A blank Voice is valid on mixed tts+clone families (plain TTS —
the model's own default voice) and, on any clone-capable entry,
when an Instructions text substitutes for the voice: on families
that condition synthesis on instructions alone the client designs
the voice from it (instruction-voice mode).
"""
+ if all_selected(fields):
+ return all_voice_problem()
if model_capability(fields) != AUDIOCPP_VOICE_CLONE:
return None
has_instruction = bool(str(_field_value(
@@ -1336,10 +1542,18 @@ def _audiocpp_fields(stdscr, api_url: Optional[str] = None,
def entry_supports_options(fs) -> bool:
"""True when the selected entry's family defines request options.
- Resolved strictly from this machine's model_specs: a family the
- specs prove unable to read options, or cannot classify at all,
- keeps the field hidden (unknown support is treated as no).
+ With "All" picked: any configured model's family with declared
+ request options shows the field (the server ignores keys a model
+ does not know). Resolved strictly from this machine's model_specs:
+ a family the specs prove unable to read options, or cannot
+ classify at all, keeps the field hidden (unknown support is
+ treated as no).
"""
+ if all_selected(fs):
+ return any(
+ audiocpp_backend.supports_request_options(
+ option_families, m.get("family") or "") is True
+ for m in models)
family = model_entry(fs).get("family") or ""
return audiocpp_backend.supports_request_options(
option_families, family) is True
@@ -1358,10 +1572,47 @@ def _audiocpp_fields(stdscr, api_url: Optional[str] = None,
"Examples: emotion=neutral, speed=1.1, temperature=0.8",
]
+ def all_voice_visible(fs) -> bool:
+ """Whether the Voice field applies to the current Model pick.
+
+ With "All" picked: shown when at least one configured model takes
+ a voice (the pick feeds every model that accepts it); hidden when
+ every model designs or plainly synthesizes. A single model keeps
+ the per-entry rule: hidden on design entries (the voice is
+ described) and on pure-TTS families (no cloning, no voice).
+ """
+ if all_selected(fs):
+ return any_voice_model()
+ return not (
+ model_capability(fs) == AUDIOCPP_VOICE_DESIGN
+ or (model_capability(fs) == AUDIOCPP_VOICE_CLONE
+ and model_voice_policy(fs) == AUDIOCPP_VOICE_NONE))
+
+ def instructions_validate(value) -> Optional[str]:
+ """Refuse a blank Instructions when the run needs it for a voice.
+
+ Single-model: required on design entries (the voice comes from
+ it). "All": required when any configured model is a design model,
+ even though the text is optional style control for the others.
+ """
+ if all_selected(fields):
+ if any_design_model() and not str(value).strip():
+ return ("The 'All' run includes a voice design model — "
+ "describe the voice in Instructions")
+ return None
+ if model_capability(fields) != AUDIOCPP_VOICE_DESIGN \
+ or str(value).strip():
+ return None
+ return "Describe the voice, e.g. 'A warm female narrator'"
+
fields = [
{"key": prefix + "model_id", "label": "Model", "kind": "choice",
"value": default_model,
- "choices": [(_label(m), m.get("id")) for m in models],
+ "choices": [(_label(m), m.get("id")) for m in models]
+ + ([("All (multiple generation)", AUDIOCPP_MODEL_ALL)]
+ # Offered with more than one model configured: a single-model
+ # server has nothing to compare.
+ if len(models) > 1 else []),
# The pick menu shows the padded capability table; the form row
# collapses its column padding back to the two-space gutter.
"compact_label": True,
@@ -1377,10 +1628,7 @@ def _audiocpp_fields(stdscr, api_url: Optional[str] = None,
"kind": "choice",
"value": initial_voice,
"choices": lambda fs: voice_choices(fs),
- "visible": lambda fs: not (
- model_capability(fs) == AUDIOCPP_VOICE_DESIGN
- or (model_capability(fs) == AUDIOCPP_VOICE_CLONE
- and model_voice_policy(fs) == AUDIOCPP_VOICE_NONE)),
+ "visible": lambda fs: all_voice_visible(fs),
"on_empty_choices": no_voices_hint,
"validate": voice_validate},
# Style/voice-design instruction. Required for design entries (the
@@ -1390,9 +1638,7 @@ def _audiocpp_fields(stdscr, api_url: Optional[str] = None,
{"key": prefix + "instructions", "label": "Instructions", "kind": "text",
"value": "",
"help": INSTRUCTIONS_HELP,
- "validate": lambda value: None
- if (model_capability(fields) != AUDIOCPP_VOICE_DESIGN or str(value).strip())
- else "Describe the voice, e.g. 'A warm female narrator'"},
+ "validate": instructions_validate},
# Free-form per-model controls (--option KEY=VALUE on the CLI).
# Shown only for families whose audio.cpp spec declares request
# options; unknown-support families keep it hidden.
@@ -1404,10 +1650,6 @@ def _audiocpp_fields(stdscr, api_url: Optional[str] = None,
]
def mapper(result) -> Optional[tuple]:
- model_id = result[prefix + "model_id"]
- # The picked voice (a built-in speaker name on a CustomVoice entry,
- # a server-side preset otherwise); the client resolves which it is.
- voice = result[prefix + "audiocpp_voice"] or None
# The instruction is forwarded for every capability: required on
# design entries, optional style/delivery control elsewhere. With
# no voice it defines the voice on instruction-conditioned families.
@@ -1419,7 +1661,6 @@ def _audiocpp_fields(stdscr, api_url: Optional[str] = None,
except ValueError:
request_options = {} # submit-time validation already caught this
kwargs = {
- "model_id": model_id, "voice": voice,
"instructions": instructions,
"request_options": request_options,
**_common_kwargs(result),
@@ -1432,6 +1673,23 @@ def _audiocpp_fields(stdscr, api_url: Optional[str] = None,
kwargs["audiocpp_rehost"] = True
else:
kwargs["api_url"] = api_url
+ if result[prefix + "model_id"] == AUDIOCPP_MODEL_ALL:
+ # "All (multiple generation)": one conversion per configured
+ # model, each with the picked voice where the model accepts
+ # it and its per-model fallback where it does not
+ # (see all_voice_for). audiobook.convert unloads loaded
+ # models between the per-model conversions.
+ picked = result.get(prefix + "audiocpp_voice") or ""
+ kwargs["model_ids"] = [m.get("id") for m in models]
+ kwargs["model_voices"] = {
+ model_id: all_voice_for(model_id, picked)
+ for model_id in kwargs["model_ids"]}
+ return ("convert", BACKEND_AUDIOCPP, kwargs)
+ model_id = result[prefix + "model_id"]
+ # The picked voice (a built-in speaker name on a CustomVoice entry,
+ # a server-side preset otherwise); the client resolves which it is.
+ kwargs["model_id"] = model_id
+ kwargs["voice"] = result[prefix + "audiocpp_voice"] or None
return ("convert", BACKEND_AUDIOCPP, kwargs)
return fields, mapper
diff --git a/app/ui/runview.py b/app/ui/runview.py
index 19a8f64..759fcfb 100644
--- a/app/ui/runview.py
+++ b/app/ui/runview.py
@@ -155,11 +155,11 @@ class RunView(ScreenView):
self.server = "starting"
self.server_message = ""
self.log_tail: List[str] = []
- self.book: Optional[tuple] = None # (index, total, name)
+ self.book: Optional[tuple] = None # (index, total, name, model)
self.chapter: Optional[tuple] = None # (index, total)
self.chunk_done = 0
self.chunk_total = 0
- self.book_results: List[tuple] = [] # (name, ok)
+ self.book_results: List[tuple] = [] # (name, ok, files, error, model)
self.error_message = ""
self.started_server = False # cancelled/cancelling/finished_at: base
self.boot_started: Optional[float] = None
@@ -211,7 +211,7 @@ class RunView(ScreenView):
elif kind == "book":
self.phase = "convert"
self.book = (event.get("index"), event.get("total"),
- event.get("name") or "")
+ event.get("name") or "", event.get("model"))
self.chapter = None
self.chunk_done = 0
self.chunk_total = 0
@@ -239,11 +239,12 @@ class RunView(ScreenView):
self.book_results.append((event.get("name") or "?",
bool(event.get("ok")),
list(event.get("files") or []),
- ""))
+ "", event.get("model")))
elif kind == "book_failed":
self.book_results.append((event.get("name") or "?", False,
list(event.get("files") or []),
- event.get("error") or "conversion failed"))
+ event.get("error") or "conversion failed",
+ event.get("model")))
self.error_message = self.error_message or \
(event.get("error") or "conversion failed")
elif kind == "done":
@@ -486,9 +487,10 @@ class RunView(ScreenView):
lines = ["Audiobook generation finished",
f"Output directory: {AUDIOBOOKS_FOLDER}"]
ok_count = 0
- for name, ok, files, error in self.book_results:
+ for name, ok, files, error, model in self.book_results:
ok_count += 1 if ok else 0
- lines.append(f"{'[OK]' if ok else '[FAIL]'} {name}"
+ label = f"{name} — {model}" if model else name
+ lines.append(f"{'[OK]' if ok else '[FAIL]'} {label}"
+ (f": {', '.join(files)}" if files else ""))
if not ok and error:
lines.append(f" {error}")
@@ -583,10 +585,13 @@ class RunView(ScreenView):
def _draw_progress(self, scr, theme, y, inner_x, label_w, value_x,
value_w, width) -> int:
"""The live panel: book, chapter, chunk bar, elapsed, message."""
- # Book line
+ # Book line ("— model" appended while an "All" run generates with
+ # a specific model, e.g. "1/6 dune.epub — qwen3_tts_..._q8_0")
if self.book is not None:
- index, total, name = self.book
+ index, total, name, model = self.book
book_text = f"{index}/{total} {name}"
+ if model:
+ book_text += f" — {model}"
else:
book_text = "waiting..." if self.phase == "convert" else "-"
_text(scr, theme, y, inner_x, "Book".ljust(label_w), theme["dim"])
@@ -660,10 +665,11 @@ class RunView(ScreenView):
_text(scr, theme, y, value_x, _fit(result, value_w),
theme.get(kind, theme["body"]))
y += 1
- for name, ok, _files, _error in self.book_results[:5]:
+ for name, ok, _files, _error, model in self.book_results[:5]:
mark = "[OK] " if ok else "[FAIL]"
+ label = f"{name} — {model}" if model else name
_text(scr, theme, y, value_x,
- _fit(f"{mark} {name}", value_w),
+ _fit(f"{mark} {label}", value_w),
theme["ok"] if ok else theme["err"])
y += 1
if len(self.book_results) > 5: