aboutsummaryrefslogtreecommitdiff
path: root/app/converter/converter.py
diff options
context:
space:
mode:
authorhistoria <historiavg@proton.me>2026-08-24 02:59:26 -0400
committerhistoria <historiavg@proton.me>2026-08-24 02:59:26 -0400
commitf00249db9d1ea051d29aa1bcca869fc4b88e83eb (patch)
treea75f076fac1b63e0b4bf2eb8f54affbcc681a891 /app/converter/converter.py
parent9dd4f9595be3b1d76a3a07dc3eca90cfaf8a3f97 (diff)
downloadtts-audiobook-generator-f00249db9d1ea051d29aa1bcca869fc4b88e83eb.tar.gz
refactor: add app directory, dir structure change
Diffstat (limited to 'app/converter/converter.py')
-rw-r--r--app/converter/converter.py782
1 files changed, 782 insertions, 0 deletions
diff --git a/app/converter/converter.py b/app/converter/converter.py
new file mode 100644
index 0000000..cef4808
--- /dev/null
+++ b/app/converter/converter.py
@@ -0,0 +1,782 @@
+"""Orchestrates book-to-audiobook conversion."""
+
+import glob
+import logging
+import re
+import shutil
+import sys
+import time
+import traceback
+from collections import Counter
+from datetime import datetime
+from pathlib import Path
+from typing import Dict, List, Optional, Tuple
+
+from . import audio, chunking, config, cover, extractors
+from .audio import TrackMeta
+from .tts import (
+ BACKENDS,
+ BACKEND_AUDIOCPP,
+ BACKEND_FASTER,
+ BACKEND_QWEN,
+ MODEL_SIZE,
+ VOICE_MODE_CLONE,
+ VOICE_MODE_CUSTOM,
+ VOICE_MODES,
+ AudioCppTTSClient,
+ FasterTTSClient,
+ QwenTTSClient,
+ normalize_language,
+ speaker_display_name,
+)
+
+logger = logging.getLogger(__name__)
+
+# Folders, resolved from the project root so the converter runs from any
+# working directory. User-facing dirs (input/, output/) stay at the root;
+# scratch/log dirs live under the app/ container.
+BASE_DIR = Path(__file__).resolve().parent.parent.parent
+APP_DIR = BASE_DIR / "app"
+
+BOOKS_FOLDER = BASE_DIR / "input"
+AUDIOBOOKS_FOLDER = BASE_DIR / "output"
+CHUNKS_FOLDER = APP_DIR / "chunks" # Per-chunk scratch audio, cleaned per book
+LOGS_FOLDER = APP_DIR / "logs"
+DEBUG_FOLDER = APP_DIR / "debug" # --debug dumps, kept across runs
+
+# Output containers and supported input formats.
+AUDIO_FORMATS = ("mp3", "m4b", "ogg", "flac")
+SUPPORTED_FORMATS = [".txt", ".pdf", ".epub"]
+
+
+def _console_log_filter(record: logging.LogRecord) -> bool:
+ """Keep httpx/httpcore request logs out of the console (file only)."""
+ return not record.name.startswith(("httpx", "httpcore"))
+
+
+def setup_logging(debug: bool = False) -> None:
+ """Configure logging to a dated file and the console.
+
+ The file keeps the full record (DEBUG with --debug), including httpx
+ request logs. The console handler only surfaces warnings and errors
+ (DEBUG with --debug) so progress prints are never mirrored as
+ timestamped log lines; httpx/httpcore request logs stay file-only.
+ """
+ LOGS_FOLDER.mkdir(parents=True, exist_ok=True)
+ file_handler = logging.FileHandler(
+ LOGS_FOLDER / f"audiobook_{datetime.now():%Y%m%d}.log",
+ encoding="utf-8",
+ )
+ file_handler.setLevel(logging.DEBUG if debug else logging.INFO)
+ console_handler = logging.StreamHandler(sys.stdout)
+ console_handler.setLevel(logging.DEBUG if debug else logging.WARNING)
+ console_handler.addFilter(_console_log_filter)
+ logging.basicConfig(
+ level=logging.INFO,
+ format="%(asctime)s - %(levelname)s - %(message)s",
+ handlers=[file_handler, console_handler],
+ )
+ if debug:
+ logging.getLogger("converter").setLevel(logging.DEBUG)
+
+
+def setup_directories() -> None:
+ """Create necessary directories."""
+ for directory in (BOOKS_FOLDER, AUDIOBOOKS_FOLDER,
+ CHUNKS_FOLDER, LOGS_FOLDER):
+ Path(directory).mkdir(parents=True, exist_ok=True)
+
+
+def find_existing_outputs(output_name: str, output_format: str) -> List[Path]:
+ """Return existing output files that a conversion would overwrite.
+
+ Multi-section books (e.g. EPUB chapters) and speed-adjusted copies are
+ named ``{name}_suffix.{ext}``; exact chapter file names are only known
+ after text extraction, so any file matching that pattern counts.
+ """
+ folder = AUDIOBOOKS_FOLDER
+ existing: List[Path] = []
+ primary = folder / f"{output_name}.{output_format}"
+ if primary.exists():
+ existing.append(primary)
+ existing.extend(sorted(
+ folder.glob(f"{glob.escape(output_name)}_*.{output_format}")))
+ return existing
+
+
+def prompt_overwrite(existing: List[Path], output_name: str) -> bool:
+ """Ask whether to reconvert a book whose output files already exist.
+
+ All overwrite questions are asked before any conversion starts so the
+ rest of the run is unattended. Pressing Enter defaults to yes (so a
+ user can just hit Enter through the prompts), but a closed stdin
+ (non-interactive run) declines and keeps existing files safe.
+ """
+ if len(existing) == 1:
+ message = f"{existing[0].name} already exists. Convert anyway and overwrite it?"
+ else:
+ message = (f"{len(existing)} output files for '{output_name}' already exist "
+ f"(e.g. {existing[0].name}). Convert anyway and overwrite them?")
+ while True:
+ try:
+ answer = input(f"{message} [Y/n]: ").strip().lower()
+ except EOFError:
+ print("\n[WARNING] No interactive input available; keeping existing output")
+ return False
+ if not answer:
+ return True
+ if answer in ("y", "yes"):
+ return True
+ if answer in ("n", "no"):
+ return False
+ print("Please answer 'y' or 'n' (or press Enter for yes).")
+
+
+class AudiobookConverter:
+ """Audiobook converter using a local TTS API."""
+
+ def __init__(self, voice_mode: str = VOICE_MODE_CUSTOM, voice_clone_ref_audio: Optional[str] = None,
+ voice_clone_ref_text: Optional[str] = None, skip_transcription: bool = False,
+ speed: float = 1.0, single_file: bool = False, output_format: str = config.AUDIO_FORMAT,
+ language: Optional[str] = None, backend: str = config.BACKEND,
+ voice: Optional[str] = None, debug: bool = False,
+ chunk: bool = False, model_id: Optional[str] = None,
+ instructions: Optional[str] = None,
+ request_options: Optional[Dict[str, str]] = None):
+ if speed <= 0:
+ raise ValueError(f"Speed must be a positive number, got {speed}")
+ if output_format not in AUDIO_FORMATS:
+ raise ValueError(f"Unsupported output format: {output_format}")
+ if backend not in BACKENDS:
+ raise ValueError(
+ f"Unknown backend: {backend!r} (expected one of {BACKENDS})"
+ )
+ if language is None:
+ language = config.LANGUAGE
+ self.language = normalize_language(language)
+ self.voice_mode = voice_mode
+ self.voice_clone_ref_audio = voice_clone_ref_audio
+ self.speed = speed
+ self.single_file = single_file
+ self.output_format = output_format
+ self.backend = backend
+ self.voice = voice
+ self.debug = bool(debug)
+ # Client-side chunking: the qwen and faster backends always chunk
+ # (their servers do one generation per request and silently truncate
+ # long text). The audio.cpp server chunks long text itself, so it
+ # defaults to one request per chapter; --chunk forces client-side
+ # chunking on top (possible needless double-chunking).
+ self.client_chunks = bool(chunk) or backend != BACKEND_AUDIOCPP
+ # Voice design / style instruction and free-form request options
+ # (audio.cpp only): forwarded to AudioCppTTSClient, which validates
+ # them against the server-hosted model at connect time.
+ self.instructions = instructions
+ self.request_options = dict(request_options or {})
+ self._validate_configuration()
+ if backend == BACKEND_FASTER:
+ # The faster backend always voice-clones using a reference voice
+ # configured on the server, so no local reference audio is needed.
+ self.tts = FasterTTSClient(voice=voice)
+ elif backend == BACKEND_AUDIOCPP:
+ # Speaker mode (no voice) uses a built-in CustomVoice speaker;
+ # an explicit voice selects a server-side preset (cloning).
+ # model_id overrides AUDIOCPP_MODEL_ID for multi-model servers;
+ # instructions describe or style the voice, request_options pass
+ # per-model controls through to the server.
+ self.tts = AudioCppTTSClient(voice=voice, language=self.language,
+ chunk_text=self.client_chunks,
+ model_id=model_id,
+ instructions=instructions,
+ request_options=self.request_options)
+ else:
+ self.tts = QwenTTSClient(
+ voice_mode=voice_mode,
+ voice_clone_ref_audio=voice_clone_ref_audio,
+ voice_clone_ref_text=voice_clone_ref_text,
+ skip_transcription=skip_transcription,
+ language=self.language,
+ )
+
+ def _validate_configuration(self) -> None:
+ """Validate configuration settings."""
+ if self.voice_mode not in VOICE_MODES:
+ raise ValueError(
+ f"Unknown voice mode: {self.voice_mode!r} "
+ f"(expected one of {VOICE_MODES})"
+ )
+ if self.voice_mode == VOICE_MODE_CLONE and self.backend == BACKEND_QWEN:
+ if not self.voice_clone_ref_audio:
+ raise ValueError(
+ "Voice Clone mode requires a reference audio file. "
+ "Use --clone <path> to specify it."
+ )
+
+ if not Path(self.voice_clone_ref_audio).exists():
+ raise ValueError(
+ f"Reference audio file not found: {self.voice_clone_ref_audio}"
+ )
+
+ @staticmethod
+ def _sanitize_filename(name: str, fallback: str = "chapter") -> str:
+ """Make a chapter title safe to use as part of a file name."""
+ cleaned = re.sub(r'[\\/:*?"<>|]', " ", name)
+ cleaned = re.sub(r"\s+", " ", cleaned).strip().strip(".")
+ return cleaned[:80] or fallback
+
+ def _narrator_tag(self) -> str:
+ """Narrator name used in output file names (see compute_narrator_tag)."""
+ return self.compute_narrator_tag(
+ self.backend, self.voice, self.voice_mode,
+ self.voice_clone_ref_audio, self.instructions)
+
+ @staticmethod
+ def compute_narrator_tag(backend: str, voice: Optional[str],
+ voice_mode: str,
+ voice_clone_ref_audio: Optional[str],
+ instructions: Optional[str] = None) -> str:
+ """Narrator name used in output file names, without a server connection.
+
+ Custom voice mode uses the built-in speaker's display name; voice
+ clone mode uses the reference audio file's stem; the faster and
+ audiocpp backends use the server-side voice name (falling back to
+ the built-in speaker for the audiocpp backend's speaker mode). An
+ instruction without a voice (voice design, or instruction-defined
+ voices on families without built-in speakers) uses "designed".
+ Spaces become underscores (e.g. "Uncle Fu" -> "Uncle_Fu").
+
+ Pure (no I/O, no server) so the pre-flight overwrite check can
+ compute the exact output names a run would produce before spending
+ time connecting to a TTS server.
+ """
+ if backend == BACKEND_FASTER:
+ narrator = voice or config.FASTER_VOICE
+ elif backend == BACKEND_AUDIOCPP:
+ if voice:
+ narrator = voice
+ elif instructions:
+ # The voice comes from the instruction, not a speaker name.
+ narrator = "designed"
+ else:
+ narrator = speaker_display_name()
+ elif voice_mode == VOICE_MODE_CLONE:
+ narrator = Path(voice_clone_ref_audio).stem
+ else:
+ narrator = speaker_display_name()
+ return AudiobookConverter._sanitize_filename(
+ narrator, fallback="narrator").replace(" ", "_")
+
+ # ------------------------------------------------------------------
+ # Debug dumps (--debug)
+ # ------------------------------------------------------------------
+
+ @staticmethod
+ def _write_debug_text(debug_dir: Path, chunk_num: int, text: str) -> None:
+ """Write the exact text sent for a chunk to the debug folder.
+
+ Called before the request so the text survives a crash mid-generation.
+ A failed debug write must never abort a conversion.
+ """
+ try:
+ debug_dir.mkdir(parents=True, exist_ok=True)
+ (debug_dir / f"chunk_{chunk_num:04d}.txt").write_text(text, encoding="utf-8")
+ except OSError as exc:
+ logger.warning("Could not write debug text for chunk %d: %s", chunk_num, exc)
+
+ @staticmethod
+ def _copy_debug_audio(debug_dir: Path, chunk_num: int, source: Path) -> Optional[Path]:
+ """Copy a generated chunk's audio file into the debug folder.
+
+ Returns the copy's path, or None when the copy failed (which never
+ affects the conversion itself).
+ """
+ try:
+ debug_dir.mkdir(parents=True, exist_ok=True)
+ target = debug_dir / f"chunk_{chunk_num:04d}{source.suffix or '.wav'}"
+ shutil.copy2(source, target)
+ return target
+ except OSError as exc:
+ logger.warning("Could not write debug audio for chunk %d: %s", chunk_num, exc)
+ return None
+
+ @staticmethod
+ def _chapter_debug_dir(book_debug_dir: Optional[Path], index: int, title: str) -> Optional[Path]:
+ """Per-chapter subfolder of a book's debug folder (None when not debugging).
+
+ Chunk numbering restarts for each chapter, so chapters get their own
+ subfolder (e.g. debug/dune_Vivian/03_The_Trial/).
+ """
+ if book_debug_dir is None:
+ return None
+ return book_debug_dir / f"{index:02d}_{AudiobookConverter._sanitize_filename(title)}"
+
+ def convert_book(self, file_path: Path, output_name: Optional[str] = None) -> bool:
+ """Convert a single book to one or more audiobook files."""
+ logger.info("Converting: %s", file_path.name)
+ start_time = time.time()
+
+ try:
+ # Start from a clean scratch folder so a previous crash can never
+ # affect this run
+ audio.cleanup_chunks()
+
+ logger.info("Extracting text...")
+ book = extractors.extract_book(file_path)
+ sections = book.sections
+ if not sections or all(not s.text.strip() for s in sections):
+ logger.error("No text extracted")
+ return False
+
+ stem = output_name or f"{file_path.stem}_{self._narrator_tag()}"
+
+ # --debug: chunk text/audio dumps land in a per-book folder
+ debug_dir = DEBUG_FOLDER / stem if self.debug else None
+
+ # Cover art: generated once per book. Named with the chunk_
+ # prefix so cleanup_chunks() removes it with the other scratch
+ # files at the end of the book.
+ cover_path = cover.generate_cover(
+ book.title, CHUNKS_FOLDER / "chunk_cover.png")
+ if cover_path:
+ print(f"[INFO] Generated cover art for '{book.title}'")
+ meta = TrackMeta(title=book.title, artist=book.author, album=book.title)
+
+ # m4b is always a single file; multi-chapter books get embedded
+ # chapter markers so listeners can skip between chapters.
+ if self.output_format == "m4b":
+ if len(sections) > 1:
+ return self._convert_m4b_with_chapters(sections, stem, start_time,
+ meta=meta, cover=cover_path,
+ debug_dir=debug_dir)
+ output_path = AUDIOBOOKS_FOLDER / f"{stem}.{self.output_format}"
+ return self._convert_text(sections[0].text, output_path, start_time,
+ meta=meta, cover=cover_path, debug_dir=debug_dir)
+
+ if self.single_file or len(sections) == 1:
+ text = "\n\n".join(section.text for section in sections)
+ output_path = AUDIOBOOKS_FOLDER / f"{stem}.{self.output_format}"
+ return self._convert_text(text, output_path, start_time,
+ meta=meta, cover=cover_path, debug_dir=debug_dir)
+
+ success = True
+ for index, section in enumerate(sections, 1):
+ chapter_name = f"{stem}_{index:02d}_{self._sanitize_filename(section.title)}"
+ output_path = AUDIOBOOKS_FOLDER / f"{chapter_name}.{self.output_format}"
+ track_meta = meta._replace(
+ title=(section.title or "").strip() or f"Chapter {index}",
+ track=index, total_tracks=len(sections))
+ success = self._convert_text(
+ section.text, output_path, time.time(),
+ meta=track_meta, cover=cover_path,
+ debug_dir=self._chapter_debug_dir(debug_dir, index, section.title)
+ ) and success
+ return success
+
+ except Exception as exc:
+ logger.error("Conversion failed: %s", exc)
+ logger.error(traceback.format_exc())
+ return False
+ finally:
+ # Always cleanup, even on failure or interrupt
+ audio.cleanup_chunks()
+
+ def _convert_m4b_with_chapters(self, sections, stem: str, start_time: float,
+ meta: Optional[TrackMeta] = None,
+ cover: Optional[Path] = None,
+ debug_dir: Optional[Path] = None) -> bool:
+ """Convert each chapter to audio, then assemble a single m4b with
+ embedded chapter markers.
+
+ Chapters are synthesized to lossless WAV scratch files (~170 MB per
+ hour of audio) so the final AAC pass is the only lossy encode. When
+ ``debug_dir`` is given, each chapter's debug dumps land in its own
+ subfolder (chunk numbering restarts per chapter).
+ """
+ chapter_files = []
+ titles = []
+ total_chapters = len(sections)
+ for index, section in enumerate(sections, 1):
+ chapter_path = CHUNKS_FOLDER / f"chapter_{index:04d}.wav"
+ title = (section.title or "").strip() or f"Chapter {index}"
+ print(f"\n{'=' * 50}")
+ print(f"CHAPTER {index}/{total_chapters}: {title}")
+ print(f"{'=' * 50}")
+ logger.info("Converting chapter %d/%d: %s", index, total_chapters, title)
+ if not self._convert_text(section.text, chapter_path, time.time(),
+ speed=1.0, output_format="wav",
+ chapter=(index, total_chapters),
+ debug_dir=self._chapter_debug_dir(debug_dir, index, title)):
+ logger.error("Chapter %d (%s) failed; aborting the conversion",
+ index, title)
+ return False
+ chapter_files.append(chapter_path)
+ titles.append(title)
+
+ if not chapter_files:
+ logger.error("No chapters were successfully converted")
+ return False
+
+ output_path = AUDIOBOOKS_FOLDER / f"{stem}.{self.output_format}"
+ if not audio.combine_chapters_to_m4b(chapter_files, titles, output_path, speed=self.speed,
+ meta=meta, cover=cover):
+ return False
+ duration = time.time() - start_time
+ logger.info("Conversion completed in %dm %ds: %s",
+ int(duration // 60), int(duration % 60), output_path)
+ return True
+
+ def _synthesize_chunks(self, chunks: List[str],
+ debug_dir: Optional[Path] = None) -> Dict[int, Optional[Path]]:
+ """Synthesize chunks sequentially, preserving order and naming.
+
+ Returns a mapping of chunk number to the generated audio path, with
+ None for chunks that failed. Generation stops at the first failed
+ chunk: a partial audiobook is never assembled, so the remaining
+ chunks are not requested. When ``debug_dir`` is given (--debug),
+ each chunk's request text and returned audio are also dumped there,
+ and every request/response is logged.
+ """
+ total_chunks = len(chunks)
+ if self.client_chunks:
+ print(f"\n{'=' * 50}")
+ print(f"PROCESSING {total_chunks} CHUNKS")
+ print(f"{'=' * 50}")
+
+ results: Dict[int, Optional[Path]] = {}
+ for chunk_num, chunk_text in enumerate(chunks, 1):
+ if debug_dir is not None:
+ # Written before the request so the exact text survives a
+ # crash mid-generation; failed chunks keep their dumps.
+ self._write_debug_text(debug_dir, chunk_num, chunk_text)
+ logger.debug("Chunk %d/%d request text: %s", chunk_num, total_chunks, chunk_text)
+ request_start = time.time()
+ try:
+ result = self.tts.process_chunk_with_retry(chunk_num, chunk_text)
+ results[chunk_num] = result
+
+ if result:
+ if debug_dir is not None:
+ copied = self._copy_debug_audio(debug_dir, chunk_num, Path(result))
+ elapsed = time.time() - request_start
+ destination = f" -> {copied.name}" if copied else ""
+ logger.debug("Chunk %d/%d response in %.1fs%s",
+ chunk_num, total_chunks, elapsed, destination)
+ if self.client_chunks:
+ print(f"[OK] Chunk {chunk_num:3d}/{total_chunks} completed")
+ logger.info("+ Chunk %d/%d completed", chunk_num, total_chunks)
+ else:
+ logger.error("Chunk %d/%d failed; aborting the remaining chunks",
+ chunk_num, total_chunks)
+ break
+
+ except Exception as exc:
+ results[chunk_num] = None
+ logger.error("Chunk %d/%d error: %s; aborting the remaining chunks",
+ chunk_num, total_chunks, exc)
+ break
+
+ successful_chunks = sum(1 for path in results.values() if path)
+ if self.client_chunks:
+ print(f"\n{'=' * 50}")
+ print("CHUNK PROCESSING COMPLETE")
+ print(f"Successful: {successful_chunks}/{total_chunks}")
+ print(f"{'=' * 50}")
+ logger.info("Chunk processing completed: %d/%d chunks", successful_chunks, total_chunks)
+ return results
+
+ def _chapter_chunks(self, text: str) -> List[str]:
+ """Split chapter text into TTS requests.
+
+ Client-side chunking splits into CHUNK_SIZE-word chunks (qwen and
+ faster always; audio.cpp only with --chunk). Otherwise (audio.cpp
+ default) the whole text is one request and the server does its own
+ long-form chunking.
+ """
+ if self.client_chunks:
+ return chunking.split_into_chunks(text)
+ return [text] if text.strip() else []
+
+ def _convert_text(self, text: str, output_path: Path, start_time: float,
+ speed: Optional[float] = None,
+ output_format: Optional[str] = None,
+ chapter: Optional[Tuple[int, int]] = None,
+ meta: Optional[TrackMeta] = None,
+ cover: Optional[Path] = None,
+ debug_dir: Optional[Path] = None) -> bool:
+ """Chunk, synthesize, and assemble ``text`` into ``output_path``.
+
+ When ``chapter`` (a ``(number, total)`` pair) is given, the output is
+ an intermediate per-chapter file and progress messages are phrased
+ accordingly instead of implying the whole book is done. ``debug_dir``
+ (from --debug) receives the chunks' text and audio dumps.
+ """
+ if speed is None:
+ speed = self.speed
+ if output_format is None:
+ output_format = self.output_format
+
+ try:
+ if not text.strip():
+ logger.error("No text to convert for %s", output_path.name)
+ return False
+
+ logger.info("Extracted %d characters (%d words)", len(text), len(text.split()))
+
+ chunks = self._chapter_chunks(text)
+ total_chunks = len(chunks)
+ if total_chunks == 0:
+ logger.error("No chunks created")
+ return False
+
+ chunk_sizes = [len(chunk.split()) for chunk in chunks]
+ avg_chunk_size = sum(chunk_sizes) / len(chunk_sizes)
+ if len(chunks) == 1:
+ logger.info("Sending the whole text as one request (%d words; "
+ "the server chunks long text itself)",
+ chunk_sizes[0])
+ else:
+ logger.info("Split into %d chunks (avg %.0f words per chunk)",
+ total_chunks, avg_chunk_size)
+ backend_labels = {
+ BACKEND_FASTER: "faster TTS API",
+ BACKEND_AUDIOCPP: "audio.cpp server",
+ }
+ backend = backend_labels.get(self.backend, "Qwen API")
+ if self.client_chunks:
+ print(f"[INFO] Processing {total_chunks} chunks via {backend}...")
+ else:
+ # The whole request is sent at once and the server does its
+ # own long-form chunking, so the chunk vocabulary does not
+ # apply; warn that this one request can take a very long time.
+ subject = (f"chapter {chapter[0]}/{chapter[1]}"
+ if chapter is not None else "text")
+ print(f"[INFO] Sending the {subject} to the {backend} as a "
+ "single request...")
+ print("[NOTE] It is expected for this to take a very long "
+ "time: the server synthesizes the entire request before "
+ "returning any audio.")
+
+ results = self._synthesize_chunks(chunks, debug_dir=debug_dir)
+ successful_chunks = sum(1 for path in results.values() if path)
+
+ if successful_chunks < total_chunks:
+ logger.error("Chunk processing incomplete (%d/%d chunks); "
+ "aborting without producing an audiobook",
+ successful_chunks, total_chunks)
+ return False
+
+ success = audio.combine_chunks(total_chunks, output_path, chunk_results=results,
+ speed=speed, output_format=output_format,
+ intermediate=chapter is not None,
+ meta=meta, cover=cover)
+
+ if success:
+ duration = time.time() - start_time
+ minutes = int(duration // 60)
+ seconds = int(duration % 60)
+ if chapter is not None:
+ logger.info("Chapter %d/%d converted in %dm %ds (%d/%d chunks)",
+ chapter[0], chapter[1], minutes, seconds,
+ successful_chunks, total_chunks)
+ if self.client_chunks:
+ print(f"[INFO] Chapter {chapter[0]}/{chapter[1]} converted "
+ f"({successful_chunks}/{total_chunks} chunks)")
+ else:
+ print(f"[INFO] Chapter {chapter[0]}/{chapter[1]} converted")
+ else:
+ logger.info("Conversion completed in %dm %ds: %s", minutes, seconds, output_path)
+ else:
+ logger.error("Failed to combine chunks into final audiobook")
+
+ return success
+
+ except Exception as exc:
+ logger.error("Conversion failed: %s", exc)
+ logger.error(traceback.format_exc())
+ return False
+
+ def _print_banner(self) -> None:
+ """Print the startup summary for the selected backend."""
+ print("=" * 70)
+ print("TTS AUDIOBOOK GENERATOR")
+ print("=" * 70)
+ print(f"Books folder: {BOOKS_FOLDER}")
+ print(f"Output folder: {AUDIOBOOKS_FOLDER}")
+ if self.backend == BACKEND_FASTER:
+ print(f"Faster TTS endpoint: {config.FASTER_API_URL}")
+ print("Backend: faster (voice cloning, reference configured on server)")
+ print(f"Voice: {self.voice or config.FASTER_VOICE}")
+ elif self.backend == BACKEND_AUDIOCPP:
+ print(f"audio.cpp endpoint: {config.AUDIOCPP_API_URL}")
+ print(f"Model id: {self.tts.model_id}")
+ print(f"Model family: {getattr(self.tts, 'family', 'unknown')}")
+ if self.voice:
+ print("Backend: audio.cpp (voice cloning, reference configured on server)")
+ print(f"Voice: {self.voice}")
+ elif self.instructions:
+ print("Backend: audio.cpp (voice from --instructions description)")
+ print(f"Instruction: {self.instructions}")
+ else:
+ print("Backend: audio.cpp (custom voice, built-in speaker)")
+ print(f"Speaker: {config.SPEAKER}")
+ if self.request_options:
+ print(f"Request options: {self.request_options}")
+ if self.client_chunks:
+ print("Chunking: client-side (--chunk; the server also chunks "
+ "long text itself, so this may double-chunk)")
+ else:
+ print("Chunking: server-side (one request per chapter; "
+ "--chunk forces client-side chunking)")
+ print(f"Language: {self.language}")
+ else:
+ api_url = (config.CLONE_API_URL if self.voice_mode == VOICE_MODE_CLONE
+ else config.QWEN_API_URL)
+ print(f"Qwen API endpoint: {api_url}")
+ print(f"Voice mode: {self.voice_mode}")
+ print(f"Model size: {MODEL_SIZE} (always)")
+ if self.voice_mode == VOICE_MODE_CUSTOM:
+ print(f"Speaker: {config.SPEAKER}")
+ print(f"Language: {self.language}")
+ elif self.voice_mode == VOICE_MODE_CLONE:
+ print(f"Reference audio: {Path(self.voice_clone_ref_audio).name}")
+ print(f"Language: {self.language}")
+ print(f"Output format: {self.output_format}")
+ if self.single_file and self.output_format != "m4b":
+ print("Chapter mode: single file (--single-file)")
+ if abs(self.speed - 1.0) >= 1e-6:
+ print(f"Playback speed: {self.speed:g}x")
+ if self.debug:
+ print(f"Debug dumps (per-chunk text + raw audio): {DEBUG_FOLDER}")
+ print("=" * 70)
+
+ # ------------------------------------------------------------------
+ # Pre-flight: overwrite checks before connecting to a TTS server
+ # ------------------------------------------------------------------
+
+ @staticmethod
+ def preflight_overwrites(backend: str, voice: Optional[str],
+ voice_mode: str,
+ voice_clone_ref_audio: Optional[str],
+ output_format: str,
+ instructions: Optional[str] = None
+ ) -> Tuple[List[Path], List[Tuple[Path, str]]]:
+ """Discover books and ask every overwrite question up front.
+
+ Pure of the TTS server: it scans the books folder, computes the
+ output name each book would produce (including the narrator tag
+ and stem-collision suffix), and asks whether to overwrite any
+ existing output files. Returns ``(book_files, planned)`` where
+ ``planned`` is the subset the user agreed to (re)convert.
+
+ Asking before connecting means a user who declines a prompt (or has
+ nothing to convert) never waits on a slow server handshake.
+ """
+ book_files = sorted(
+ f for f in BOOKS_FOLDER.iterdir()
+ if f.is_file() and f.suffix.lower() in SUPPORTED_FORMATS
+ )
+ if not book_files:
+ return [], []
+
+ print(f"[INFO] Found {len(book_files)} books to convert")
+
+ # Avoid output collisions when two books share a stem (e.g. dune.txt + dune.epub).
+ stem_counts: Dict[str, int] = Counter(book_file.stem for book_file in book_files)
+
+ # Ask every overwrite question up front, before any conversion
+ # starts, so the rest of the run is unattended.
+ planned: List[Tuple[Path, str]] = []
+ narrator_tag = AudiobookConverter.compute_narrator_tag(
+ backend, voice, voice_mode, voice_clone_ref_audio, instructions)
+ for book_file in book_files:
+ output_name = book_file.stem
+ if stem_counts[book_file.stem] > 1:
+ output_name = f"{book_file.stem}_{book_file.suffix.lstrip('.')}"
+ output_name = f"{output_name}_{narrator_tag}"
+ existing = find_existing_outputs(output_name, output_format)
+ if existing and not prompt_overwrite(existing, output_name):
+ print(f"[INFO] Skipping {book_file.name} (existing output kept)")
+ continue
+ planned.append((book_file, output_name))
+ return book_files, planned
+
+ # ------------------------------------------------------------------
+ # Main conversion loop
+ # ------------------------------------------------------------------
+
+ def run(self) -> bool:
+ """Main conversion process. Returns True if all books converted."""
+ run_start = time.time()
+ self._print_banner()
+
+ # When main() has already done the pre-flight overwrite check, use
+ # its results so the prompts are not asked a second time; otherwise
+ # (e.g. a converter constructed directly) discover and ask here.
+ if getattr(self, "_planned", None) is not None:
+ book_files = self._book_files
+ planned = self._planned
+ else:
+ book_files, planned = AudiobookConverter.preflight_overwrites(
+ self.backend, self.voice, self.voice_mode,
+ self.voice_clone_ref_audio, self.output_format,
+ self.instructions)
+
+ if not book_files:
+ print(f"[INFO] No supported files found in {BOOKS_FOLDER}")
+ print(f"Supported formats: {', '.join(SUPPORTED_FORMATS)}")
+ print("[INFO] Nothing to convert. Add a .txt, .pdf, or .epub file "
+ f"to {BOOKS_FOLDER} and run again.")
+ return True
+
+ if not planned:
+ print("[INFO] Nothing to convert (all books skipped)")
+ return True
+
+ print(f"[INFO] Converting {len(planned)} of {len(book_files)} book(s)")
+
+ results = {}
+ for book_file, output_name in planned:
+ try:
+ success = self.convert_book(book_file, output_name=output_name)
+ results[book_file.name] = success
+ except KeyboardInterrupt:
+ print("\n[WARNING] Conversion interrupted by user")
+ results[book_file.name] = False
+ break
+ except Exception as exc:
+ logger.error("Unexpected error: %s", exc)
+ results[book_file.name] = False
+ if not results[book_file.name]:
+ logger.error("Conversion of %s failed; aborting the remaining books",
+ book_file.name)
+ break
+
+ successful = sum(results.values())
+ total = len(results)
+
+ print("\n" + "=" * 70)
+ print("CONVERSION SUMMARY")
+ print("=" * 70)
+ print(f"Total: {total} | Success: {successful} | Failed: {total - successful}")
+ print("=" * 70)
+
+ for filename, success in results.items():
+ status = "[OK]" if success else "[FAIL]"
+ print(f"{status} {filename}")
+
+ if successful > 0:
+ print(f"\n[INFO] Audiobooks saved to: {AUDIOBOOKS_FOLDER}/")
+
+ elapsed = int(time.time() - run_start)
+ hours, remainder = divmod(elapsed, 3600)
+ minutes, seconds = divmod(remainder, 60)
+ if hours:
+ duration = f"{hours}h {minutes}m {seconds}s"
+ elif minutes:
+ duration = f"{minutes}m {seconds}s"
+ else:
+ duration = f"{seconds}s"
+ print(f"\n[INFO] Generation completed in {duration}")
+ logger.info("Generation completed in %s", duration)
+
+ return total > 0 and successful == total