aboutsummaryrefslogtreecommitdiff
path: root/converter/converter.py
diff options
context:
space:
mode:
authorhistoria <historiavg@proton.me>2026-08-24 02:59:26 -0400
committerhistoria <historiavg@proton.me>2026-08-24 02:59:26 -0400
commitf00249db9d1ea051d29aa1bcca869fc4b88e83eb (patch)
treea75f076fac1b63e0b4bf2eb8f54affbcc681a891 /converter/converter.py
parent9dd4f9595be3b1d76a3a07dc3eca90cfaf8a3f97 (diff)
downloadtts-audiobook-generator-f00249db9d1ea051d29aa1bcca869fc4b88e83eb.tar.gz
refactor: add app directory, dir structure change
Diffstat (limited to 'converter/converter.py')
-rw-r--r--converter/converter.py780
1 files changed, 0 insertions, 780 deletions
diff --git a/converter/converter.py b/converter/converter.py
deleted file mode 100644
index b24ac09..0000000
--- a/converter/converter.py
+++ /dev/null
@@ -1,780 +0,0 @@
-"""Orchestrates book-to-audiobook conversion."""
-
-import glob
-import logging
-import re
-import shutil
-import sys
-import time
-import traceback
-from collections import Counter
-from datetime import datetime
-from pathlib import Path
-from typing import Dict, List, Optional, Tuple
-
-from . import audio, chunking, config, cover, extractors
-from .audio import TrackMeta
-from .tts import (
- BACKENDS,
- BACKEND_AUDIOCPP,
- BACKEND_FASTER,
- BACKEND_QWEN,
- MODEL_SIZE,
- VOICE_MODE_CLONE,
- VOICE_MODE_CUSTOM,
- VOICE_MODES,
- AudioCppTTSClient,
- FasterTTSClient,
- QwenTTSClient,
- normalize_language,
- speaker_display_name,
-)
-
-logger = logging.getLogger(__name__)
-
-# Folders, resolved from the project root so the converter runs from any
-# working directory.
-BASE_DIR = Path(__file__).resolve().parent.parent
-
-BOOKS_FOLDER = BASE_DIR / "input"
-AUDIOBOOKS_FOLDER = BASE_DIR / "output"
-CHUNKS_FOLDER = BASE_DIR / "chunks" # Per-chunk scratch audio, cleaned per book
-LOGS_FOLDER = BASE_DIR / "logs"
-DEBUG_FOLDER = BASE_DIR / "debug" # --debug dumps, kept across runs
-
-# Output containers and supported input formats.
-AUDIO_FORMATS = ("mp3", "m4b", "ogg", "flac")
-SUPPORTED_FORMATS = [".txt", ".pdf", ".epub"]
-
-
-def _console_log_filter(record: logging.LogRecord) -> bool:
- """Keep httpx/httpcore request logs out of the console (file only)."""
- return not record.name.startswith(("httpx", "httpcore"))
-
-
-def setup_logging(debug: bool = False) -> None:
- """Configure logging to a dated file and the console.
-
- The file keeps the full record (DEBUG with --debug), including httpx
- request logs. The console handler only surfaces warnings and errors
- (DEBUG with --debug) so progress prints are never mirrored as
- timestamped log lines; httpx/httpcore request logs stay file-only.
- """
- LOGS_FOLDER.mkdir(parents=True, exist_ok=True)
- file_handler = logging.FileHandler(
- LOGS_FOLDER / f"audiobook_{datetime.now():%Y%m%d}.log",
- encoding="utf-8",
- )
- file_handler.setLevel(logging.DEBUG if debug else logging.INFO)
- console_handler = logging.StreamHandler(sys.stdout)
- console_handler.setLevel(logging.DEBUG if debug else logging.WARNING)
- console_handler.addFilter(_console_log_filter)
- logging.basicConfig(
- level=logging.INFO,
- format="%(asctime)s - %(levelname)s - %(message)s",
- handlers=[file_handler, console_handler],
- )
- if debug:
- logging.getLogger("converter").setLevel(logging.DEBUG)
-
-
-def setup_directories() -> None:
- """Create necessary directories."""
- for directory in (BOOKS_FOLDER, AUDIOBOOKS_FOLDER,
- CHUNKS_FOLDER, LOGS_FOLDER):
- Path(directory).mkdir(parents=True, exist_ok=True)
-
-
-def find_existing_outputs(output_name: str, output_format: str) -> List[Path]:
- """Return existing output files that a conversion would overwrite.
-
- Multi-section books (e.g. EPUB chapters) and speed-adjusted copies are
- named ``{name}_suffix.{ext}``; exact chapter file names are only known
- after text extraction, so any file matching that pattern counts.
- """
- folder = AUDIOBOOKS_FOLDER
- existing: List[Path] = []
- primary = folder / f"{output_name}.{output_format}"
- if primary.exists():
- existing.append(primary)
- existing.extend(sorted(
- folder.glob(f"{glob.escape(output_name)}_*.{output_format}")))
- return existing
-
-
-def prompt_overwrite(existing: List[Path], output_name: str) -> bool:
- """Ask whether to reconvert a book whose output files already exist.
-
- All overwrite questions are asked before any conversion starts so the
- rest of the run is unattended. Pressing Enter defaults to yes (so a
- user can just hit Enter through the prompts), but a closed stdin
- (non-interactive run) declines and keeps existing files safe.
- """
- if len(existing) == 1:
- message = f"{existing[0].name} already exists. Convert anyway and overwrite it?"
- else:
- message = (f"{len(existing)} output files for '{output_name}' already exist "
- f"(e.g. {existing[0].name}). Convert anyway and overwrite them?")
- while True:
- try:
- answer = input(f"{message} [Y/n]: ").strip().lower()
- except EOFError:
- print("\n[WARNING] No interactive input available; keeping existing output")
- return False
- if not answer:
- return True
- if answer in ("y", "yes"):
- return True
- if answer in ("n", "no"):
- return False
- print("Please answer 'y' or 'n' (or press Enter for yes).")
-
-
-class AudiobookConverter:
- """Audiobook converter using a local TTS API."""
-
- def __init__(self, voice_mode: str = VOICE_MODE_CUSTOM, voice_clone_ref_audio: Optional[str] = None,
- voice_clone_ref_text: Optional[str] = None, skip_transcription: bool = False,
- speed: float = 1.0, single_file: bool = False, output_format: str = config.AUDIO_FORMAT,
- language: Optional[str] = None, backend: str = config.BACKEND,
- voice: Optional[str] = None, debug: bool = False,
- chunk: bool = False, model_id: Optional[str] = None,
- instructions: Optional[str] = None,
- request_options: Optional[Dict[str, str]] = None):
- if speed <= 0:
- raise ValueError(f"Speed must be a positive number, got {speed}")
- if output_format not in AUDIO_FORMATS:
- raise ValueError(f"Unsupported output format: {output_format}")
- if backend not in BACKENDS:
- raise ValueError(
- f"Unknown backend: {backend!r} (expected one of {BACKENDS})"
- )
- if language is None:
- language = config.LANGUAGE
- self.language = normalize_language(language)
- self.voice_mode = voice_mode
- self.voice_clone_ref_audio = voice_clone_ref_audio
- self.speed = speed
- self.single_file = single_file
- self.output_format = output_format
- self.backend = backend
- self.voice = voice
- self.debug = bool(debug)
- # Client-side chunking: the qwen and faster backends always chunk
- # (their servers do one generation per request and silently truncate
- # long text). The audio.cpp server chunks long text itself, so it
- # defaults to one request per chapter; --chunk forces client-side
- # chunking on top (possible needless double-chunking).
- self.client_chunks = bool(chunk) or backend != BACKEND_AUDIOCPP
- # Voice design / style instruction and free-form request options
- # (audio.cpp only): forwarded to AudioCppTTSClient, which validates
- # them against the server-hosted model at connect time.
- self.instructions = instructions
- self.request_options = dict(request_options or {})
- self._validate_configuration()
- if backend == BACKEND_FASTER:
- # The faster backend always voice-clones using a reference voice
- # configured on the server, so no local reference audio is needed.
- self.tts = FasterTTSClient(voice=voice)
- elif backend == BACKEND_AUDIOCPP:
- # Speaker mode (no voice) uses a built-in CustomVoice speaker;
- # an explicit voice selects a server-side preset (cloning).
- # model_id overrides AUDIOCPP_MODEL_ID for multi-model servers;
- # instructions describe or style the voice, request_options pass
- # per-model controls through to the server.
- self.tts = AudioCppTTSClient(voice=voice, language=self.language,
- chunk_text=self.client_chunks,
- model_id=model_id,
- instructions=instructions,
- request_options=self.request_options)
- else:
- self.tts = QwenTTSClient(
- voice_mode=voice_mode,
- voice_clone_ref_audio=voice_clone_ref_audio,
- voice_clone_ref_text=voice_clone_ref_text,
- skip_transcription=skip_transcription,
- language=self.language,
- )
-
- def _validate_configuration(self) -> None:
- """Validate configuration settings."""
- if self.voice_mode not in VOICE_MODES:
- raise ValueError(
- f"Unknown voice mode: {self.voice_mode!r} "
- f"(expected one of {VOICE_MODES})"
- )
- if self.voice_mode == VOICE_MODE_CLONE and self.backend == BACKEND_QWEN:
- if not self.voice_clone_ref_audio:
- raise ValueError(
- "Voice Clone mode requires a reference audio file. "
- "Use --clone <path> to specify it."
- )
-
- if not Path(self.voice_clone_ref_audio).exists():
- raise ValueError(
- f"Reference audio file not found: {self.voice_clone_ref_audio}"
- )
-
- @staticmethod
- def _sanitize_filename(name: str, fallback: str = "chapter") -> str:
- """Make a chapter title safe to use as part of a file name."""
- cleaned = re.sub(r'[\\/:*?"<>|]', " ", name)
- cleaned = re.sub(r"\s+", " ", cleaned).strip().strip(".")
- return cleaned[:80] or fallback
-
- def _narrator_tag(self) -> str:
- """Narrator name used in output file names (see compute_narrator_tag)."""
- return self.compute_narrator_tag(
- self.backend, self.voice, self.voice_mode,
- self.voice_clone_ref_audio, self.instructions)
-
- @staticmethod
- def compute_narrator_tag(backend: str, voice: Optional[str],
- voice_mode: str,
- voice_clone_ref_audio: Optional[str],
- instructions: Optional[str] = None) -> str:
- """Narrator name used in output file names, without a server connection.
-
- Custom voice mode uses the built-in speaker's display name; voice
- clone mode uses the reference audio file's stem; the faster and
- audiocpp backends use the server-side voice name (falling back to
- the built-in speaker for the audiocpp backend's speaker mode). An
- instruction without a voice (voice design, or instruction-defined
- voices on families without built-in speakers) uses "designed".
- Spaces become underscores (e.g. "Uncle Fu" -> "Uncle_Fu").
-
- Pure (no I/O, no server) so the pre-flight overwrite check can
- compute the exact output names a run would produce before spending
- time connecting to a TTS server.
- """
- if backend == BACKEND_FASTER:
- narrator = voice or config.FASTER_VOICE
- elif backend == BACKEND_AUDIOCPP:
- if voice:
- narrator = voice
- elif instructions:
- # The voice comes from the instruction, not a speaker name.
- narrator = "designed"
- else:
- narrator = speaker_display_name()
- elif voice_mode == VOICE_MODE_CLONE:
- narrator = Path(voice_clone_ref_audio).stem
- else:
- narrator = speaker_display_name()
- return AudiobookConverter._sanitize_filename(
- narrator, fallback="narrator").replace(" ", "_")
-
- # ------------------------------------------------------------------
- # Debug dumps (--debug)
- # ------------------------------------------------------------------
-
- @staticmethod
- def _write_debug_text(debug_dir: Path, chunk_num: int, text: str) -> None:
- """Write the exact text sent for a chunk to the debug folder.
-
- Called before the request so the text survives a crash mid-generation.
- A failed debug write must never abort a conversion.
- """
- try:
- debug_dir.mkdir(parents=True, exist_ok=True)
- (debug_dir / f"chunk_{chunk_num:04d}.txt").write_text(text, encoding="utf-8")
- except OSError as exc:
- logger.warning("Could not write debug text for chunk %d: %s", chunk_num, exc)
-
- @staticmethod
- def _copy_debug_audio(debug_dir: Path, chunk_num: int, source: Path) -> Optional[Path]:
- """Copy a generated chunk's audio file into the debug folder.
-
- Returns the copy's path, or None when the copy failed (which never
- affects the conversion itself).
- """
- try:
- debug_dir.mkdir(parents=True, exist_ok=True)
- target = debug_dir / f"chunk_{chunk_num:04d}{source.suffix or '.wav'}"
- shutil.copy2(source, target)
- return target
- except OSError as exc:
- logger.warning("Could not write debug audio for chunk %d: %s", chunk_num, exc)
- return None
-
- @staticmethod
- def _chapter_debug_dir(book_debug_dir: Optional[Path], index: int, title: str) -> Optional[Path]:
- """Per-chapter subfolder of a book's debug folder (None when not debugging).
-
- Chunk numbering restarts for each chapter, so chapters get their own
- subfolder (e.g. debug/dune_Vivian/03_The_Trial/).
- """
- if book_debug_dir is None:
- return None
- return book_debug_dir / f"{index:02d}_{AudiobookConverter._sanitize_filename(title)}"
-
- def convert_book(self, file_path: Path, output_name: Optional[str] = None) -> bool:
- """Convert a single book to one or more audiobook files."""
- logger.info("Converting: %s", file_path.name)
- start_time = time.time()
-
- try:
- # Start from a clean scratch folder so a previous crash can never
- # affect this run
- audio.cleanup_chunks()
-
- logger.info("Extracting text...")
- book = extractors.extract_book(file_path)
- sections = book.sections
- if not sections or all(not s.text.strip() for s in sections):
- logger.error("No text extracted")
- return False
-
- stem = output_name or f"{file_path.stem}_{self._narrator_tag()}"
-
- # --debug: chunk text/audio dumps land in a per-book folder
- debug_dir = DEBUG_FOLDER / stem if self.debug else None
-
- # Cover art: generated once per book. Named with the chunk_
- # prefix so cleanup_chunks() removes it with the other scratch
- # files at the end of the book.
- cover_path = cover.generate_cover(
- book.title, CHUNKS_FOLDER / "chunk_cover.png")
- if cover_path:
- print(f"[INFO] Generated cover art for '{book.title}'")
- meta = TrackMeta(title=book.title, artist=book.author, album=book.title)
-
- # m4b is always a single file; multi-chapter books get embedded
- # chapter markers so listeners can skip between chapters.
- if self.output_format == "m4b":
- if len(sections) > 1:
- return self._convert_m4b_with_chapters(sections, stem, start_time,
- meta=meta, cover=cover_path,
- debug_dir=debug_dir)
- output_path = AUDIOBOOKS_FOLDER / f"{stem}.{self.output_format}"
- return self._convert_text(sections[0].text, output_path, start_time,
- meta=meta, cover=cover_path, debug_dir=debug_dir)
-
- if self.single_file or len(sections) == 1:
- text = "\n\n".join(section.text for section in sections)
- output_path = AUDIOBOOKS_FOLDER / f"{stem}.{self.output_format}"
- return self._convert_text(text, output_path, start_time,
- meta=meta, cover=cover_path, debug_dir=debug_dir)
-
- success = True
- for index, section in enumerate(sections, 1):
- chapter_name = f"{stem}_{index:02d}_{self._sanitize_filename(section.title)}"
- output_path = AUDIOBOOKS_FOLDER / f"{chapter_name}.{self.output_format}"
- track_meta = meta._replace(
- title=(section.title or "").strip() or f"Chapter {index}",
- track=index, total_tracks=len(sections))
- success = self._convert_text(
- section.text, output_path, time.time(),
- meta=track_meta, cover=cover_path,
- debug_dir=self._chapter_debug_dir(debug_dir, index, section.title)
- ) and success
- return success
-
- except Exception as exc:
- logger.error("Conversion failed: %s", exc)
- logger.error(traceback.format_exc())
- return False
- finally:
- # Always cleanup, even on failure or interrupt
- audio.cleanup_chunks()
-
- def _convert_m4b_with_chapters(self, sections, stem: str, start_time: float,
- meta: Optional[TrackMeta] = None,
- cover: Optional[Path] = None,
- debug_dir: Optional[Path] = None) -> bool:
- """Convert each chapter to audio, then assemble a single m4b with
- embedded chapter markers.
-
- Chapters are synthesized to lossless WAV scratch files (~170 MB per
- hour of audio) so the final AAC pass is the only lossy encode. When
- ``debug_dir`` is given, each chapter's debug dumps land in its own
- subfolder (chunk numbering restarts per chapter).
- """
- chapter_files = []
- titles = []
- total_chapters = len(sections)
- for index, section in enumerate(sections, 1):
- chapter_path = CHUNKS_FOLDER / f"chapter_{index:04d}.wav"
- title = (section.title or "").strip() or f"Chapter {index}"
- print(f"\n{'=' * 50}")
- print(f"CHAPTER {index}/{total_chapters}: {title}")
- print(f"{'=' * 50}")
- logger.info("Converting chapter %d/%d: %s", index, total_chapters, title)
- if not self._convert_text(section.text, chapter_path, time.time(),
- speed=1.0, output_format="wav",
- chapter=(index, total_chapters),
- debug_dir=self._chapter_debug_dir(debug_dir, index, title)):
- logger.error("Chapter %d (%s) failed; aborting the conversion",
- index, title)
- return False
- chapter_files.append(chapter_path)
- titles.append(title)
-
- if not chapter_files:
- logger.error("No chapters were successfully converted")
- return False
-
- output_path = AUDIOBOOKS_FOLDER / f"{stem}.{self.output_format}"
- if not audio.combine_chapters_to_m4b(chapter_files, titles, output_path, speed=self.speed,
- meta=meta, cover=cover):
- return False
- duration = time.time() - start_time
- logger.info("Conversion completed in %dm %ds: %s",
- int(duration // 60), int(duration % 60), output_path)
- return True
-
- def _synthesize_chunks(self, chunks: List[str],
- debug_dir: Optional[Path] = None) -> Dict[int, Optional[Path]]:
- """Synthesize chunks sequentially, preserving order and naming.
-
- Returns a mapping of chunk number to the generated audio path, with
- None for chunks that failed. Generation stops at the first failed
- chunk: a partial audiobook is never assembled, so the remaining
- chunks are not requested. When ``debug_dir`` is given (--debug),
- each chunk's request text and returned audio are also dumped there,
- and every request/response is logged.
- """
- total_chunks = len(chunks)
- if self.client_chunks:
- print(f"\n{'=' * 50}")
- print(f"PROCESSING {total_chunks} CHUNKS")
- print(f"{'=' * 50}")
-
- results: Dict[int, Optional[Path]] = {}
- for chunk_num, chunk_text in enumerate(chunks, 1):
- if debug_dir is not None:
- # Written before the request so the exact text survives a
- # crash mid-generation; failed chunks keep their dumps.
- self._write_debug_text(debug_dir, chunk_num, chunk_text)
- logger.debug("Chunk %d/%d request text: %s", chunk_num, total_chunks, chunk_text)
- request_start = time.time()
- try:
- result = self.tts.process_chunk_with_retry(chunk_num, chunk_text)
- results[chunk_num] = result
-
- if result:
- if debug_dir is not None:
- copied = self._copy_debug_audio(debug_dir, chunk_num, Path(result))
- elapsed = time.time() - request_start
- destination = f" -> {copied.name}" if copied else ""
- logger.debug("Chunk %d/%d response in %.1fs%s",
- chunk_num, total_chunks, elapsed, destination)
- if self.client_chunks:
- print(f"[OK] Chunk {chunk_num:3d}/{total_chunks} completed")
- logger.info("+ Chunk %d/%d completed", chunk_num, total_chunks)
- else:
- logger.error("Chunk %d/%d failed; aborting the remaining chunks",
- chunk_num, total_chunks)
- break
-
- except Exception as exc:
- results[chunk_num] = None
- logger.error("Chunk %d/%d error: %s; aborting the remaining chunks",
- chunk_num, total_chunks, exc)
- break
-
- successful_chunks = sum(1 for path in results.values() if path)
- if self.client_chunks:
- print(f"\n{'=' * 50}")
- print("CHUNK PROCESSING COMPLETE")
- print(f"Successful: {successful_chunks}/{total_chunks}")
- print(f"{'=' * 50}")
- logger.info("Chunk processing completed: %d/%d chunks", successful_chunks, total_chunks)
- return results
-
- def _chapter_chunks(self, text: str) -> List[str]:
- """Split chapter text into TTS requests.
-
- Client-side chunking splits into CHUNK_SIZE-word chunks (qwen and
- faster always; audio.cpp only with --chunk). Otherwise (audio.cpp
- default) the whole text is one request and the server does its own
- long-form chunking.
- """
- if self.client_chunks:
- return chunking.split_into_chunks(text)
- return [text] if text.strip() else []
-
- def _convert_text(self, text: str, output_path: Path, start_time: float,
- speed: Optional[float] = None,
- output_format: Optional[str] = None,
- chapter: Optional[Tuple[int, int]] = None,
- meta: Optional[TrackMeta] = None,
- cover: Optional[Path] = None,
- debug_dir: Optional[Path] = None) -> bool:
- """Chunk, synthesize, and assemble ``text`` into ``output_path``.
-
- When ``chapter`` (a ``(number, total)`` pair) is given, the output is
- an intermediate per-chapter file and progress messages are phrased
- accordingly instead of implying the whole book is done. ``debug_dir``
- (from --debug) receives the chunks' text and audio dumps.
- """
- if speed is None:
- speed = self.speed
- if output_format is None:
- output_format = self.output_format
-
- try:
- if not text.strip():
- logger.error("No text to convert for %s", output_path.name)
- return False
-
- logger.info("Extracted %d characters (%d words)", len(text), len(text.split()))
-
- chunks = self._chapter_chunks(text)
- total_chunks = len(chunks)
- if total_chunks == 0:
- logger.error("No chunks created")
- return False
-
- chunk_sizes = [len(chunk.split()) for chunk in chunks]
- avg_chunk_size = sum(chunk_sizes) / len(chunk_sizes)
- if len(chunks) == 1:
- logger.info("Sending the whole text as one request (%d words; "
- "the server chunks long text itself)",
- chunk_sizes[0])
- else:
- logger.info("Split into %d chunks (avg %.0f words per chunk)",
- total_chunks, avg_chunk_size)
- backend_labels = {
- BACKEND_FASTER: "faster TTS API",
- BACKEND_AUDIOCPP: "audio.cpp server",
- }
- backend = backend_labels.get(self.backend, "Qwen API")
- if self.client_chunks:
- print(f"[INFO] Processing {total_chunks} chunks via {backend}...")
- else:
- # The whole request is sent at once and the server does its
- # own long-form chunking, so the chunk vocabulary does not
- # apply; warn that this one request can take a very long time.
- subject = (f"chapter {chapter[0]}/{chapter[1]}"
- if chapter is not None else "text")
- print(f"[INFO] Sending the {subject} to the {backend} as a "
- "single request...")
- print("[NOTE] It is expected for this to take a very long "
- "time: the server synthesizes the entire request before "
- "returning any audio.")
-
- results = self._synthesize_chunks(chunks, debug_dir=debug_dir)
- successful_chunks = sum(1 for path in results.values() if path)
-
- if successful_chunks < total_chunks:
- logger.error("Chunk processing incomplete (%d/%d chunks); "
- "aborting without producing an audiobook",
- successful_chunks, total_chunks)
- return False
-
- success = audio.combine_chunks(total_chunks, output_path, chunk_results=results,
- speed=speed, output_format=output_format,
- intermediate=chapter is not None,
- meta=meta, cover=cover)
-
- if success:
- duration = time.time() - start_time
- minutes = int(duration // 60)
- seconds = int(duration % 60)
- if chapter is not None:
- logger.info("Chapter %d/%d converted in %dm %ds (%d/%d chunks)",
- chapter[0], chapter[1], minutes, seconds,
- successful_chunks, total_chunks)
- if self.client_chunks:
- print(f"[INFO] Chapter {chapter[0]}/{chapter[1]} converted "
- f"({successful_chunks}/{total_chunks} chunks)")
- else:
- print(f"[INFO] Chapter {chapter[0]}/{chapter[1]} converted")
- else:
- logger.info("Conversion completed in %dm %ds: %s", minutes, seconds, output_path)
- else:
- logger.error("Failed to combine chunks into final audiobook")
-
- return success
-
- except Exception as exc:
- logger.error("Conversion failed: %s", exc)
- logger.error(traceback.format_exc())
- return False
-
- def _print_banner(self) -> None:
- """Print the startup summary for the selected backend."""
- print("=" * 70)
- print("TTS AUDIOBOOK GENERATOR")
- print("=" * 70)
- print(f"Books folder: {BOOKS_FOLDER}")
- print(f"Output folder: {AUDIOBOOKS_FOLDER}")
- if self.backend == BACKEND_FASTER:
- print(f"Faster TTS endpoint: {config.FASTER_API_URL}")
- print("Backend: faster (voice cloning, reference configured on server)")
- print(f"Voice: {self.voice or config.FASTER_VOICE}")
- elif self.backend == BACKEND_AUDIOCPP:
- print(f"audio.cpp endpoint: {config.AUDIOCPP_API_URL}")
- print(f"Model id: {self.tts.model_id}")
- print(f"Model family: {getattr(self.tts, 'family', 'unknown')}")
- if self.voice:
- print("Backend: audio.cpp (voice cloning, reference configured on server)")
- print(f"Voice: {self.voice}")
- elif self.instructions:
- print("Backend: audio.cpp (voice from --instructions description)")
- print(f"Instruction: {self.instructions}")
- else:
- print("Backend: audio.cpp (custom voice, built-in speaker)")
- print(f"Speaker: {config.SPEAKER}")
- if self.request_options:
- print(f"Request options: {self.request_options}")
- if self.client_chunks:
- print("Chunking: client-side (--chunk; the server also chunks "
- "long text itself, so this may double-chunk)")
- else:
- print("Chunking: server-side (one request per chapter; "
- "--chunk forces client-side chunking)")
- print(f"Language: {self.language}")
- else:
- api_url = (config.CLONE_API_URL if self.voice_mode == VOICE_MODE_CLONE
- else config.QWEN_API_URL)
- print(f"Qwen API endpoint: {api_url}")
- print(f"Voice mode: {self.voice_mode}")
- print(f"Model size: {MODEL_SIZE} (always)")
- if self.voice_mode == VOICE_MODE_CUSTOM:
- print(f"Speaker: {config.SPEAKER}")
- print(f"Language: {self.language}")
- elif self.voice_mode == VOICE_MODE_CLONE:
- print(f"Reference audio: {Path(self.voice_clone_ref_audio).name}")
- print(f"Language: {self.language}")
- print(f"Output format: {self.output_format}")
- if self.single_file and self.output_format != "m4b":
- print("Chapter mode: single file (--single-file)")
- if abs(self.speed - 1.0) >= 1e-6:
- print(f"Playback speed: {self.speed:g}x")
- if self.debug:
- print(f"Debug dumps (per-chunk text + raw audio): {DEBUG_FOLDER}")
- print("=" * 70)
-
- # ------------------------------------------------------------------
- # Pre-flight: overwrite checks before connecting to a TTS server
- # ------------------------------------------------------------------
-
- @staticmethod
- def preflight_overwrites(backend: str, voice: Optional[str],
- voice_mode: str,
- voice_clone_ref_audio: Optional[str],
- output_format: str,
- instructions: Optional[str] = None
- ) -> Tuple[List[Path], List[Tuple[Path, str]]]:
- """Discover books and ask every overwrite question up front.
-
- Pure of the TTS server: it scans the books folder, computes the
- output name each book would produce (including the narrator tag
- and stem-collision suffix), and asks whether to overwrite any
- existing output files. Returns ``(book_files, planned)`` where
- ``planned`` is the subset the user agreed to (re)convert.
-
- Asking before connecting means a user who declines a prompt (or has
- nothing to convert) never waits on a slow server handshake.
- """
- book_files = sorted(
- f for f in BOOKS_FOLDER.iterdir()
- if f.is_file() and f.suffix.lower() in SUPPORTED_FORMATS
- )
- if not book_files:
- return [], []
-
- print(f"[INFO] Found {len(book_files)} books to convert")
-
- # Avoid output collisions when two books share a stem (e.g. dune.txt + dune.epub).
- stem_counts: Dict[str, int] = Counter(book_file.stem for book_file in book_files)
-
- # Ask every overwrite question up front, before any conversion
- # starts, so the rest of the run is unattended.
- planned: List[Tuple[Path, str]] = []
- narrator_tag = AudiobookConverter.compute_narrator_tag(
- backend, voice, voice_mode, voice_clone_ref_audio, instructions)
- for book_file in book_files:
- output_name = book_file.stem
- if stem_counts[book_file.stem] > 1:
- output_name = f"{book_file.stem}_{book_file.suffix.lstrip('.')}"
- output_name = f"{output_name}_{narrator_tag}"
- existing = find_existing_outputs(output_name, output_format)
- if existing and not prompt_overwrite(existing, output_name):
- print(f"[INFO] Skipping {book_file.name} (existing output kept)")
- continue
- planned.append((book_file, output_name))
- return book_files, planned
-
- # ------------------------------------------------------------------
- # Main conversion loop
- # ------------------------------------------------------------------
-
- def run(self) -> bool:
- """Main conversion process. Returns True if all books converted."""
- run_start = time.time()
- self._print_banner()
-
- # When main() has already done the pre-flight overwrite check, use
- # its results so the prompts are not asked a second time; otherwise
- # (e.g. a converter constructed directly) discover and ask here.
- if getattr(self, "_planned", None) is not None:
- book_files = self._book_files
- planned = self._planned
- else:
- book_files, planned = AudiobookConverter.preflight_overwrites(
- self.backend, self.voice, self.voice_mode,
- self.voice_clone_ref_audio, self.output_format,
- self.instructions)
-
- if not book_files:
- print(f"[INFO] No supported files found in {BOOKS_FOLDER}")
- print(f"Supported formats: {', '.join(SUPPORTED_FORMATS)}")
- print("[INFO] Nothing to convert. Add a .txt, .pdf, or .epub file "
- f"to {BOOKS_FOLDER} and run again.")
- return True
-
- if not planned:
- print("[INFO] Nothing to convert (all books skipped)")
- return True
-
- print(f"[INFO] Converting {len(planned)} of {len(book_files)} book(s)")
-
- results = {}
- for book_file, output_name in planned:
- try:
- success = self.convert_book(book_file, output_name=output_name)
- results[book_file.name] = success
- except KeyboardInterrupt:
- print("\n[WARNING] Conversion interrupted by user")
- results[book_file.name] = False
- break
- except Exception as exc:
- logger.error("Unexpected error: %s", exc)
- results[book_file.name] = False
- if not results[book_file.name]:
- logger.error("Conversion of %s failed; aborting the remaining books",
- book_file.name)
- break
-
- successful = sum(results.values())
- total = len(results)
-
- print("\n" + "=" * 70)
- print("CONVERSION SUMMARY")
- print("=" * 70)
- print(f"Total: {total} | Success: {successful} | Failed: {total - successful}")
- print("=" * 70)
-
- for filename, success in results.items():
- status = "[OK]" if success else "[FAIL]"
- print(f"{status} {filename}")
-
- if successful > 0:
- print(f"\n[INFO] Audiobooks saved to: {AUDIOBOOKS_FOLDER}/")
-
- elapsed = int(time.time() - run_start)
- hours, remainder = divmod(elapsed, 3600)
- minutes, seconds = divmod(remainder, 60)
- if hours:
- duration = f"{hours}h {minutes}m {seconds}s"
- elif minutes:
- duration = f"{minutes}m {seconds}s"
- else:
- duration = f"{seconds}s"
- print(f"\n[INFO] Generation completed in {duration}")
- logger.info("Generation completed in %s", duration)
-
- return total > 0 and successful == total