diff options
Diffstat (limited to 'app/converter/converter.py')
| -rw-r--r-- | app/converter/converter.py | 782 |
1 files changed, 782 insertions, 0 deletions
diff --git a/app/converter/converter.py b/app/converter/converter.py new file mode 100644 index 0000000..cef4808 --- /dev/null +++ b/app/converter/converter.py @@ -0,0 +1,782 @@ +"""Orchestrates book-to-audiobook conversion.""" + +import glob +import logging +import re +import shutil +import sys +import time +import traceback +from collections import Counter +from datetime import datetime +from pathlib import Path +from typing import Dict, List, Optional, Tuple + +from . import audio, chunking, config, cover, extractors +from .audio import TrackMeta +from .tts import ( + BACKENDS, + BACKEND_AUDIOCPP, + BACKEND_FASTER, + BACKEND_QWEN, + MODEL_SIZE, + VOICE_MODE_CLONE, + VOICE_MODE_CUSTOM, + VOICE_MODES, + AudioCppTTSClient, + FasterTTSClient, + QwenTTSClient, + normalize_language, + speaker_display_name, +) + +logger = logging.getLogger(__name__) + +# Folders, resolved from the project root so the converter runs from any +# working directory. User-facing dirs (input/, output/) stay at the root; +# scratch/log dirs live under the app/ container. +BASE_DIR = Path(__file__).resolve().parent.parent.parent +APP_DIR = BASE_DIR / "app" + +BOOKS_FOLDER = BASE_DIR / "input" +AUDIOBOOKS_FOLDER = BASE_DIR / "output" +CHUNKS_FOLDER = APP_DIR / "chunks" # Per-chunk scratch audio, cleaned per book +LOGS_FOLDER = APP_DIR / "logs" +DEBUG_FOLDER = APP_DIR / "debug" # --debug dumps, kept across runs + +# Output containers and supported input formats. +AUDIO_FORMATS = ("mp3", "m4b", "ogg", "flac") +SUPPORTED_FORMATS = [".txt", ".pdf", ".epub"] + + +def _console_log_filter(record: logging.LogRecord) -> bool: + """Keep httpx/httpcore request logs out of the console (file only).""" + return not record.name.startswith(("httpx", "httpcore")) + + +def setup_logging(debug: bool = False) -> None: + """Configure logging to a dated file and the console. + + The file keeps the full record (DEBUG with --debug), including httpx + request logs. The console handler only surfaces warnings and errors + (DEBUG with --debug) so progress prints are never mirrored as + timestamped log lines; httpx/httpcore request logs stay file-only. + """ + LOGS_FOLDER.mkdir(parents=True, exist_ok=True) + file_handler = logging.FileHandler( + LOGS_FOLDER / f"audiobook_{datetime.now():%Y%m%d}.log", + encoding="utf-8", + ) + file_handler.setLevel(logging.DEBUG if debug else logging.INFO) + console_handler = logging.StreamHandler(sys.stdout) + console_handler.setLevel(logging.DEBUG if debug else logging.WARNING) + console_handler.addFilter(_console_log_filter) + logging.basicConfig( + level=logging.INFO, + format="%(asctime)s - %(levelname)s - %(message)s", + handlers=[file_handler, console_handler], + ) + if debug: + logging.getLogger("converter").setLevel(logging.DEBUG) + + +def setup_directories() -> None: + """Create necessary directories.""" + for directory in (BOOKS_FOLDER, AUDIOBOOKS_FOLDER, + CHUNKS_FOLDER, LOGS_FOLDER): + Path(directory).mkdir(parents=True, exist_ok=True) + + +def find_existing_outputs(output_name: str, output_format: str) -> List[Path]: + """Return existing output files that a conversion would overwrite. + + Multi-section books (e.g. EPUB chapters) and speed-adjusted copies are + named ``{name}_suffix.{ext}``; exact chapter file names are only known + after text extraction, so any file matching that pattern counts. + """ + folder = AUDIOBOOKS_FOLDER + existing: List[Path] = [] + primary = folder / f"{output_name}.{output_format}" + if primary.exists(): + existing.append(primary) + existing.extend(sorted( + folder.glob(f"{glob.escape(output_name)}_*.{output_format}"))) + return existing + + +def prompt_overwrite(existing: List[Path], output_name: str) -> bool: + """Ask whether to reconvert a book whose output files already exist. + + All overwrite questions are asked before any conversion starts so the + rest of the run is unattended. Pressing Enter defaults to yes (so a + user can just hit Enter through the prompts), but a closed stdin + (non-interactive run) declines and keeps existing files safe. + """ + if len(existing) == 1: + message = f"{existing[0].name} already exists. Convert anyway and overwrite it?" + else: + message = (f"{len(existing)} output files for '{output_name}' already exist " + f"(e.g. {existing[0].name}). Convert anyway and overwrite them?") + while True: + try: + answer = input(f"{message} [Y/n]: ").strip().lower() + except EOFError: + print("\n[WARNING] No interactive input available; keeping existing output") + return False + if not answer: + return True + if answer in ("y", "yes"): + return True + if answer in ("n", "no"): + return False + print("Please answer 'y' or 'n' (or press Enter for yes).") + + +class AudiobookConverter: + """Audiobook converter using a local TTS API.""" + + def __init__(self, voice_mode: str = VOICE_MODE_CUSTOM, voice_clone_ref_audio: Optional[str] = None, + voice_clone_ref_text: Optional[str] = None, skip_transcription: bool = False, + speed: float = 1.0, single_file: bool = False, output_format: str = config.AUDIO_FORMAT, + language: Optional[str] = None, backend: str = config.BACKEND, + voice: Optional[str] = None, debug: bool = False, + chunk: bool = False, model_id: Optional[str] = None, + instructions: Optional[str] = None, + request_options: Optional[Dict[str, str]] = None): + if speed <= 0: + raise ValueError(f"Speed must be a positive number, got {speed}") + if output_format not in AUDIO_FORMATS: + raise ValueError(f"Unsupported output format: {output_format}") + if backend not in BACKENDS: + raise ValueError( + f"Unknown backend: {backend!r} (expected one of {BACKENDS})" + ) + if language is None: + language = config.LANGUAGE + self.language = normalize_language(language) + self.voice_mode = voice_mode + self.voice_clone_ref_audio = voice_clone_ref_audio + self.speed = speed + self.single_file = single_file + self.output_format = output_format + self.backend = backend + self.voice = voice + self.debug = bool(debug) + # Client-side chunking: the qwen and faster backends always chunk + # (their servers do one generation per request and silently truncate + # long text). The audio.cpp server chunks long text itself, so it + # defaults to one request per chapter; --chunk forces client-side + # chunking on top (possible needless double-chunking). + self.client_chunks = bool(chunk) or backend != BACKEND_AUDIOCPP + # Voice design / style instruction and free-form request options + # (audio.cpp only): forwarded to AudioCppTTSClient, which validates + # them against the server-hosted model at connect time. + self.instructions = instructions + self.request_options = dict(request_options or {}) + self._validate_configuration() + if backend == BACKEND_FASTER: + # The faster backend always voice-clones using a reference voice + # configured on the server, so no local reference audio is needed. + self.tts = FasterTTSClient(voice=voice) + elif backend == BACKEND_AUDIOCPP: + # Speaker mode (no voice) uses a built-in CustomVoice speaker; + # an explicit voice selects a server-side preset (cloning). + # model_id overrides AUDIOCPP_MODEL_ID for multi-model servers; + # instructions describe or style the voice, request_options pass + # per-model controls through to the server. + self.tts = AudioCppTTSClient(voice=voice, language=self.language, + chunk_text=self.client_chunks, + model_id=model_id, + instructions=instructions, + request_options=self.request_options) + else: + self.tts = QwenTTSClient( + voice_mode=voice_mode, + voice_clone_ref_audio=voice_clone_ref_audio, + voice_clone_ref_text=voice_clone_ref_text, + skip_transcription=skip_transcription, + language=self.language, + ) + + def _validate_configuration(self) -> None: + """Validate configuration settings.""" + if self.voice_mode not in VOICE_MODES: + raise ValueError( + f"Unknown voice mode: {self.voice_mode!r} " + f"(expected one of {VOICE_MODES})" + ) + if self.voice_mode == VOICE_MODE_CLONE and self.backend == BACKEND_QWEN: + if not self.voice_clone_ref_audio: + raise ValueError( + "Voice Clone mode requires a reference audio file. " + "Use --clone <path> to specify it." + ) + + if not Path(self.voice_clone_ref_audio).exists(): + raise ValueError( + f"Reference audio file not found: {self.voice_clone_ref_audio}" + ) + + @staticmethod + def _sanitize_filename(name: str, fallback: str = "chapter") -> str: + """Make a chapter title safe to use as part of a file name.""" + cleaned = re.sub(r'[\\/:*?"<>|]', " ", name) + cleaned = re.sub(r"\s+", " ", cleaned).strip().strip(".") + return cleaned[:80] or fallback + + def _narrator_tag(self) -> str: + """Narrator name used in output file names (see compute_narrator_tag).""" + return self.compute_narrator_tag( + self.backend, self.voice, self.voice_mode, + self.voice_clone_ref_audio, self.instructions) + + @staticmethod + def compute_narrator_tag(backend: str, voice: Optional[str], + voice_mode: str, + voice_clone_ref_audio: Optional[str], + instructions: Optional[str] = None) -> str: + """Narrator name used in output file names, without a server connection. + + Custom voice mode uses the built-in speaker's display name; voice + clone mode uses the reference audio file's stem; the faster and + audiocpp backends use the server-side voice name (falling back to + the built-in speaker for the audiocpp backend's speaker mode). An + instruction without a voice (voice design, or instruction-defined + voices on families without built-in speakers) uses "designed". + Spaces become underscores (e.g. "Uncle Fu" -> "Uncle_Fu"). + + Pure (no I/O, no server) so the pre-flight overwrite check can + compute the exact output names a run would produce before spending + time connecting to a TTS server. + """ + if backend == BACKEND_FASTER: + narrator = voice or config.FASTER_VOICE + elif backend == BACKEND_AUDIOCPP: + if voice: + narrator = voice + elif instructions: + # The voice comes from the instruction, not a speaker name. + narrator = "designed" + else: + narrator = speaker_display_name() + elif voice_mode == VOICE_MODE_CLONE: + narrator = Path(voice_clone_ref_audio).stem + else: + narrator = speaker_display_name() + return AudiobookConverter._sanitize_filename( + narrator, fallback="narrator").replace(" ", "_") + + # ------------------------------------------------------------------ + # Debug dumps (--debug) + # ------------------------------------------------------------------ + + @staticmethod + def _write_debug_text(debug_dir: Path, chunk_num: int, text: str) -> None: + """Write the exact text sent for a chunk to the debug folder. + + Called before the request so the text survives a crash mid-generation. + A failed debug write must never abort a conversion. + """ + try: + debug_dir.mkdir(parents=True, exist_ok=True) + (debug_dir / f"chunk_{chunk_num:04d}.txt").write_text(text, encoding="utf-8") + except OSError as exc: + logger.warning("Could not write debug text for chunk %d: %s", chunk_num, exc) + + @staticmethod + def _copy_debug_audio(debug_dir: Path, chunk_num: int, source: Path) -> Optional[Path]: + """Copy a generated chunk's audio file into the debug folder. + + Returns the copy's path, or None when the copy failed (which never + affects the conversion itself). + """ + try: + debug_dir.mkdir(parents=True, exist_ok=True) + target = debug_dir / f"chunk_{chunk_num:04d}{source.suffix or '.wav'}" + shutil.copy2(source, target) + return target + except OSError as exc: + logger.warning("Could not write debug audio for chunk %d: %s", chunk_num, exc) + return None + + @staticmethod + def _chapter_debug_dir(book_debug_dir: Optional[Path], index: int, title: str) -> Optional[Path]: + """Per-chapter subfolder of a book's debug folder (None when not debugging). + + Chunk numbering restarts for each chapter, so chapters get their own + subfolder (e.g. debug/dune_Vivian/03_The_Trial/). + """ + if book_debug_dir is None: + return None + return book_debug_dir / f"{index:02d}_{AudiobookConverter._sanitize_filename(title)}" + + def convert_book(self, file_path: Path, output_name: Optional[str] = None) -> bool: + """Convert a single book to one or more audiobook files.""" + logger.info("Converting: %s", file_path.name) + start_time = time.time() + + try: + # Start from a clean scratch folder so a previous crash can never + # affect this run + audio.cleanup_chunks() + + logger.info("Extracting text...") + book = extractors.extract_book(file_path) + sections = book.sections + if not sections or all(not s.text.strip() for s in sections): + logger.error("No text extracted") + return False + + stem = output_name or f"{file_path.stem}_{self._narrator_tag()}" + + # --debug: chunk text/audio dumps land in a per-book folder + debug_dir = DEBUG_FOLDER / stem if self.debug else None + + # Cover art: generated once per book. Named with the chunk_ + # prefix so cleanup_chunks() removes it with the other scratch + # files at the end of the book. + cover_path = cover.generate_cover( + book.title, CHUNKS_FOLDER / "chunk_cover.png") + if cover_path: + print(f"[INFO] Generated cover art for '{book.title}'") + meta = TrackMeta(title=book.title, artist=book.author, album=book.title) + + # m4b is always a single file; multi-chapter books get embedded + # chapter markers so listeners can skip between chapters. + if self.output_format == "m4b": + if len(sections) > 1: + return self._convert_m4b_with_chapters(sections, stem, start_time, + meta=meta, cover=cover_path, + debug_dir=debug_dir) + output_path = AUDIOBOOKS_FOLDER / f"{stem}.{self.output_format}" + return self._convert_text(sections[0].text, output_path, start_time, + meta=meta, cover=cover_path, debug_dir=debug_dir) + + if self.single_file or len(sections) == 1: + text = "\n\n".join(section.text for section in sections) + output_path = AUDIOBOOKS_FOLDER / f"{stem}.{self.output_format}" + return self._convert_text(text, output_path, start_time, + meta=meta, cover=cover_path, debug_dir=debug_dir) + + success = True + for index, section in enumerate(sections, 1): + chapter_name = f"{stem}_{index:02d}_{self._sanitize_filename(section.title)}" + output_path = AUDIOBOOKS_FOLDER / f"{chapter_name}.{self.output_format}" + track_meta = meta._replace( + title=(section.title or "").strip() or f"Chapter {index}", + track=index, total_tracks=len(sections)) + success = self._convert_text( + section.text, output_path, time.time(), + meta=track_meta, cover=cover_path, + debug_dir=self._chapter_debug_dir(debug_dir, index, section.title) + ) and success + return success + + except Exception as exc: + logger.error("Conversion failed: %s", exc) + logger.error(traceback.format_exc()) + return False + finally: + # Always cleanup, even on failure or interrupt + audio.cleanup_chunks() + + def _convert_m4b_with_chapters(self, sections, stem: str, start_time: float, + meta: Optional[TrackMeta] = None, + cover: Optional[Path] = None, + debug_dir: Optional[Path] = None) -> bool: + """Convert each chapter to audio, then assemble a single m4b with + embedded chapter markers. + + Chapters are synthesized to lossless WAV scratch files (~170 MB per + hour of audio) so the final AAC pass is the only lossy encode. When + ``debug_dir`` is given, each chapter's debug dumps land in its own + subfolder (chunk numbering restarts per chapter). + """ + chapter_files = [] + titles = [] + total_chapters = len(sections) + for index, section in enumerate(sections, 1): + chapter_path = CHUNKS_FOLDER / f"chapter_{index:04d}.wav" + title = (section.title or "").strip() or f"Chapter {index}" + print(f"\n{'=' * 50}") + print(f"CHAPTER {index}/{total_chapters}: {title}") + print(f"{'=' * 50}") + logger.info("Converting chapter %d/%d: %s", index, total_chapters, title) + if not self._convert_text(section.text, chapter_path, time.time(), + speed=1.0, output_format="wav", + chapter=(index, total_chapters), + debug_dir=self._chapter_debug_dir(debug_dir, index, title)): + logger.error("Chapter %d (%s) failed; aborting the conversion", + index, title) + return False + chapter_files.append(chapter_path) + titles.append(title) + + if not chapter_files: + logger.error("No chapters were successfully converted") + return False + + output_path = AUDIOBOOKS_FOLDER / f"{stem}.{self.output_format}" + if not audio.combine_chapters_to_m4b(chapter_files, titles, output_path, speed=self.speed, + meta=meta, cover=cover): + return False + duration = time.time() - start_time + logger.info("Conversion completed in %dm %ds: %s", + int(duration // 60), int(duration % 60), output_path) + return True + + def _synthesize_chunks(self, chunks: List[str], + debug_dir: Optional[Path] = None) -> Dict[int, Optional[Path]]: + """Synthesize chunks sequentially, preserving order and naming. + + Returns a mapping of chunk number to the generated audio path, with + None for chunks that failed. Generation stops at the first failed + chunk: a partial audiobook is never assembled, so the remaining + chunks are not requested. When ``debug_dir`` is given (--debug), + each chunk's request text and returned audio are also dumped there, + and every request/response is logged. + """ + total_chunks = len(chunks) + if self.client_chunks: + print(f"\n{'=' * 50}") + print(f"PROCESSING {total_chunks} CHUNKS") + print(f"{'=' * 50}") + + results: Dict[int, Optional[Path]] = {} + for chunk_num, chunk_text in enumerate(chunks, 1): + if debug_dir is not None: + # Written before the request so the exact text survives a + # crash mid-generation; failed chunks keep their dumps. + self._write_debug_text(debug_dir, chunk_num, chunk_text) + logger.debug("Chunk %d/%d request text: %s", chunk_num, total_chunks, chunk_text) + request_start = time.time() + try: + result = self.tts.process_chunk_with_retry(chunk_num, chunk_text) + results[chunk_num] = result + + if result: + if debug_dir is not None: + copied = self._copy_debug_audio(debug_dir, chunk_num, Path(result)) + elapsed = time.time() - request_start + destination = f" -> {copied.name}" if copied else "" + logger.debug("Chunk %d/%d response in %.1fs%s", + chunk_num, total_chunks, elapsed, destination) + if self.client_chunks: + print(f"[OK] Chunk {chunk_num:3d}/{total_chunks} completed") + logger.info("+ Chunk %d/%d completed", chunk_num, total_chunks) + else: + logger.error("Chunk %d/%d failed; aborting the remaining chunks", + chunk_num, total_chunks) + break + + except Exception as exc: + results[chunk_num] = None + logger.error("Chunk %d/%d error: %s; aborting the remaining chunks", + chunk_num, total_chunks, exc) + break + + successful_chunks = sum(1 for path in results.values() if path) + if self.client_chunks: + print(f"\n{'=' * 50}") + print("CHUNK PROCESSING COMPLETE") + print(f"Successful: {successful_chunks}/{total_chunks}") + print(f"{'=' * 50}") + logger.info("Chunk processing completed: %d/%d chunks", successful_chunks, total_chunks) + return results + + def _chapter_chunks(self, text: str) -> List[str]: + """Split chapter text into TTS requests. + + Client-side chunking splits into CHUNK_SIZE-word chunks (qwen and + faster always; audio.cpp only with --chunk). Otherwise (audio.cpp + default) the whole text is one request and the server does its own + long-form chunking. + """ + if self.client_chunks: + return chunking.split_into_chunks(text) + return [text] if text.strip() else [] + + def _convert_text(self, text: str, output_path: Path, start_time: float, + speed: Optional[float] = None, + output_format: Optional[str] = None, + chapter: Optional[Tuple[int, int]] = None, + meta: Optional[TrackMeta] = None, + cover: Optional[Path] = None, + debug_dir: Optional[Path] = None) -> bool: + """Chunk, synthesize, and assemble ``text`` into ``output_path``. + + When ``chapter`` (a ``(number, total)`` pair) is given, the output is + an intermediate per-chapter file and progress messages are phrased + accordingly instead of implying the whole book is done. ``debug_dir`` + (from --debug) receives the chunks' text and audio dumps. + """ + if speed is None: + speed = self.speed + if output_format is None: + output_format = self.output_format + + try: + if not text.strip(): + logger.error("No text to convert for %s", output_path.name) + return False + + logger.info("Extracted %d characters (%d words)", len(text), len(text.split())) + + chunks = self._chapter_chunks(text) + total_chunks = len(chunks) + if total_chunks == 0: + logger.error("No chunks created") + return False + + chunk_sizes = [len(chunk.split()) for chunk in chunks] + avg_chunk_size = sum(chunk_sizes) / len(chunk_sizes) + if len(chunks) == 1: + logger.info("Sending the whole text as one request (%d words; " + "the server chunks long text itself)", + chunk_sizes[0]) + else: + logger.info("Split into %d chunks (avg %.0f words per chunk)", + total_chunks, avg_chunk_size) + backend_labels = { + BACKEND_FASTER: "faster TTS API", + BACKEND_AUDIOCPP: "audio.cpp server", + } + backend = backend_labels.get(self.backend, "Qwen API") + if self.client_chunks: + print(f"[INFO] Processing {total_chunks} chunks via {backend}...") + else: + # The whole request is sent at once and the server does its + # own long-form chunking, so the chunk vocabulary does not + # apply; warn that this one request can take a very long time. + subject = (f"chapter {chapter[0]}/{chapter[1]}" + if chapter is not None else "text") + print(f"[INFO] Sending the {subject} to the {backend} as a " + "single request...") + print("[NOTE] It is expected for this to take a very long " + "time: the server synthesizes the entire request before " + "returning any audio.") + + results = self._synthesize_chunks(chunks, debug_dir=debug_dir) + successful_chunks = sum(1 for path in results.values() if path) + + if successful_chunks < total_chunks: + logger.error("Chunk processing incomplete (%d/%d chunks); " + "aborting without producing an audiobook", + successful_chunks, total_chunks) + return False + + success = audio.combine_chunks(total_chunks, output_path, chunk_results=results, + speed=speed, output_format=output_format, + intermediate=chapter is not None, + meta=meta, cover=cover) + + if success: + duration = time.time() - start_time + minutes = int(duration // 60) + seconds = int(duration % 60) + if chapter is not None: + logger.info("Chapter %d/%d converted in %dm %ds (%d/%d chunks)", + chapter[0], chapter[1], minutes, seconds, + successful_chunks, total_chunks) + if self.client_chunks: + print(f"[INFO] Chapter {chapter[0]}/{chapter[1]} converted " + f"({successful_chunks}/{total_chunks} chunks)") + else: + print(f"[INFO] Chapter {chapter[0]}/{chapter[1]} converted") + else: + logger.info("Conversion completed in %dm %ds: %s", minutes, seconds, output_path) + else: + logger.error("Failed to combine chunks into final audiobook") + + return success + + except Exception as exc: + logger.error("Conversion failed: %s", exc) + logger.error(traceback.format_exc()) + return False + + def _print_banner(self) -> None: + """Print the startup summary for the selected backend.""" + print("=" * 70) + print("TTS AUDIOBOOK GENERATOR") + print("=" * 70) + print(f"Books folder: {BOOKS_FOLDER}") + print(f"Output folder: {AUDIOBOOKS_FOLDER}") + if self.backend == BACKEND_FASTER: + print(f"Faster TTS endpoint: {config.FASTER_API_URL}") + print("Backend: faster (voice cloning, reference configured on server)") + print(f"Voice: {self.voice or config.FASTER_VOICE}") + elif self.backend == BACKEND_AUDIOCPP: + print(f"audio.cpp endpoint: {config.AUDIOCPP_API_URL}") + print(f"Model id: {self.tts.model_id}") + print(f"Model family: {getattr(self.tts, 'family', 'unknown')}") + if self.voice: + print("Backend: audio.cpp (voice cloning, reference configured on server)") + print(f"Voice: {self.voice}") + elif self.instructions: + print("Backend: audio.cpp (voice from --instructions description)") + print(f"Instruction: {self.instructions}") + else: + print("Backend: audio.cpp (custom voice, built-in speaker)") + print(f"Speaker: {config.SPEAKER}") + if self.request_options: + print(f"Request options: {self.request_options}") + if self.client_chunks: + print("Chunking: client-side (--chunk; the server also chunks " + "long text itself, so this may double-chunk)") + else: + print("Chunking: server-side (one request per chapter; " + "--chunk forces client-side chunking)") + print(f"Language: {self.language}") + else: + api_url = (config.CLONE_API_URL if self.voice_mode == VOICE_MODE_CLONE + else config.QWEN_API_URL) + print(f"Qwen API endpoint: {api_url}") + print(f"Voice mode: {self.voice_mode}") + print(f"Model size: {MODEL_SIZE} (always)") + if self.voice_mode == VOICE_MODE_CUSTOM: + print(f"Speaker: {config.SPEAKER}") + print(f"Language: {self.language}") + elif self.voice_mode == VOICE_MODE_CLONE: + print(f"Reference audio: {Path(self.voice_clone_ref_audio).name}") + print(f"Language: {self.language}") + print(f"Output format: {self.output_format}") + if self.single_file and self.output_format != "m4b": + print("Chapter mode: single file (--single-file)") + if abs(self.speed - 1.0) >= 1e-6: + print(f"Playback speed: {self.speed:g}x") + if self.debug: + print(f"Debug dumps (per-chunk text + raw audio): {DEBUG_FOLDER}") + print("=" * 70) + + # ------------------------------------------------------------------ + # Pre-flight: overwrite checks before connecting to a TTS server + # ------------------------------------------------------------------ + + @staticmethod + def preflight_overwrites(backend: str, voice: Optional[str], + voice_mode: str, + voice_clone_ref_audio: Optional[str], + output_format: str, + instructions: Optional[str] = None + ) -> Tuple[List[Path], List[Tuple[Path, str]]]: + """Discover books and ask every overwrite question up front. + + Pure of the TTS server: it scans the books folder, computes the + output name each book would produce (including the narrator tag + and stem-collision suffix), and asks whether to overwrite any + existing output files. Returns ``(book_files, planned)`` where + ``planned`` is the subset the user agreed to (re)convert. + + Asking before connecting means a user who declines a prompt (or has + nothing to convert) never waits on a slow server handshake. + """ + book_files = sorted( + f for f in BOOKS_FOLDER.iterdir() + if f.is_file() and f.suffix.lower() in SUPPORTED_FORMATS + ) + if not book_files: + return [], [] + + print(f"[INFO] Found {len(book_files)} books to convert") + + # Avoid output collisions when two books share a stem (e.g. dune.txt + dune.epub). + stem_counts: Dict[str, int] = Counter(book_file.stem for book_file in book_files) + + # Ask every overwrite question up front, before any conversion + # starts, so the rest of the run is unattended. + planned: List[Tuple[Path, str]] = [] + narrator_tag = AudiobookConverter.compute_narrator_tag( + backend, voice, voice_mode, voice_clone_ref_audio, instructions) + for book_file in book_files: + output_name = book_file.stem + if stem_counts[book_file.stem] > 1: + output_name = f"{book_file.stem}_{book_file.suffix.lstrip('.')}" + output_name = f"{output_name}_{narrator_tag}" + existing = find_existing_outputs(output_name, output_format) + if existing and not prompt_overwrite(existing, output_name): + print(f"[INFO] Skipping {book_file.name} (existing output kept)") + continue + planned.append((book_file, output_name)) + return book_files, planned + + # ------------------------------------------------------------------ + # Main conversion loop + # ------------------------------------------------------------------ + + def run(self) -> bool: + """Main conversion process. Returns True if all books converted.""" + run_start = time.time() + self._print_banner() + + # When main() has already done the pre-flight overwrite check, use + # its results so the prompts are not asked a second time; otherwise + # (e.g. a converter constructed directly) discover and ask here. + if getattr(self, "_planned", None) is not None: + book_files = self._book_files + planned = self._planned + else: + book_files, planned = AudiobookConverter.preflight_overwrites( + self.backend, self.voice, self.voice_mode, + self.voice_clone_ref_audio, self.output_format, + self.instructions) + + if not book_files: + print(f"[INFO] No supported files found in {BOOKS_FOLDER}") + print(f"Supported formats: {', '.join(SUPPORTED_FORMATS)}") + print("[INFO] Nothing to convert. Add a .txt, .pdf, or .epub file " + f"to {BOOKS_FOLDER} and run again.") + return True + + if not planned: + print("[INFO] Nothing to convert (all books skipped)") + return True + + print(f"[INFO] Converting {len(planned)} of {len(book_files)} book(s)") + + results = {} + for book_file, output_name in planned: + try: + success = self.convert_book(book_file, output_name=output_name) + results[book_file.name] = success + except KeyboardInterrupt: + print("\n[WARNING] Conversion interrupted by user") + results[book_file.name] = False + break + except Exception as exc: + logger.error("Unexpected error: %s", exc) + results[book_file.name] = False + if not results[book_file.name]: + logger.error("Conversion of %s failed; aborting the remaining books", + book_file.name) + break + + successful = sum(results.values()) + total = len(results) + + print("\n" + "=" * 70) + print("CONVERSION SUMMARY") + print("=" * 70) + print(f"Total: {total} | Success: {successful} | Failed: {total - successful}") + print("=" * 70) + + for filename, success in results.items(): + status = "[OK]" if success else "[FAIL]" + print(f"{status} {filename}") + + if successful > 0: + print(f"\n[INFO] Audiobooks saved to: {AUDIOBOOKS_FOLDER}/") + + elapsed = int(time.time() - run_start) + hours, remainder = divmod(elapsed, 3600) + minutes, seconds = divmod(remainder, 60) + if hours: + duration = f"{hours}h {minutes}m {seconds}s" + elif minutes: + duration = f"{minutes}m {seconds}s" + else: + duration = f"{seconds}s" + print(f"\n[INFO] Generation completed in {duration}") + logger.info("Generation completed in %s", duration) + + return total > 0 and successful == total |
