"""Orchestrates book-to-audiobook conversion.""" import glob import logging import re import shutil import sys import time import traceback from collections import Counter from datetime import datetime from pathlib import Path from typing import Dict, List, Optional, Tuple from . import audio, chunking, config, cover, extractors from .audio import TrackMeta from .tts import FasterTTSClient, QwenTTSClient, normalize_language, speaker_display_name logger = logging.getLogger(__name__) def setup_logging(debug: bool = False) -> None: """Configure logging to both a dated file and the console.""" config.LOGS_FOLDER.mkdir(parents=True, exist_ok=True) logging.basicConfig( level=logging.INFO, format="%(asctime)s - %(levelname)s - %(message)s", handlers=[ logging.FileHandler( config.LOGS_FOLDER / f"audiobook_{datetime.now():%Y%m%d}.log", encoding="utf-8", ), logging.StreamHandler(sys.stdout), ], ) if debug: logging.getLogger("converter").setLevel(logging.DEBUG) def setup_directories() -> None: """Create necessary directories.""" for directory in (config.BOOKS_FOLDER, config.AUDIOBOOKS_FOLDER, config.CHUNKS_FOLDER, config.LOGS_FOLDER): Path(directory).mkdir(parents=True, exist_ok=True) def find_existing_outputs(output_name: str, output_format: str) -> List[Path]: """Return existing output files that a conversion would overwrite. Multi-section books (e.g. EPUB chapters) and speed-adjusted copies are named ``{name}_suffix.{ext}``; exact chapter file names are only known after text extraction, so any file matching that pattern counts. """ folder = config.AUDIOBOOKS_FOLDER existing: List[Path] = [] primary = folder / f"{output_name}.{output_format}" if primary.exists(): existing.append(primary) existing.extend(sorted( folder.glob(f"{glob.escape(output_name)}_*.{output_format}"))) return existing def prompt_overwrite(existing: List[Path], output_name: str) -> bool: """Ask whether to reconvert a book whose output files already exist. All overwrite questions are asked before any conversion starts so the rest of the run is unattended. Returns False when no interactive input is available (stdin closed), keeping existing files safe. """ if len(existing) == 1: message = f"{existing[0].name} already exists. Convert anyway and overwrite it?" else: message = (f"{len(existing)} output files for '{output_name}' already exist " f"(e.g. {existing[0].name}). Convert anyway and overwrite them?") while True: try: answer = input(f"{message} (y/n): ").strip().lower() except EOFError: print("\n[WARNING] No interactive input available; keeping existing output") return False if answer in ("y", "yes"): return True if answer in ("n", "no"): return False print("Please answer 'y' or 'n'.") class AudiobookConverter: """Audiobook converter using the Qwen TTS API.""" def __init__(self, voice_mode: str = config.VOICE_MODE_CUSTOM, voice_clone_ref_audio: Optional[str] = None, voice_clone_ref_text: Optional[str] = None, skip_transcription: bool = False, speed: float = 1.0, single_file: bool = False, output_format: str = config.AUDIO_FORMAT, language: Optional[str] = None, faster: bool = False, faster_voice: Optional[str] = None, debug: bool = False): if speed <= 0: raise ValueError(f"Speed must be a positive number, got {speed}") if output_format not in config.AUDIO_FORMATS: raise ValueError(f"Unsupported output format: {output_format}") if language is None: language = (config.VOICE_CLONE_LANGUAGE if voice_mode == config.VOICE_MODE_CLONE else config.CUSTOM_VOICE_LANGUAGE) self.language = normalize_language(language) self.voice_mode = voice_mode self.voice_clone_ref_audio = voice_clone_ref_audio self.speed = speed self.single_file = single_file self.output_format = output_format self.faster = faster self.faster_voice = faster_voice self.debug = bool(debug) self._validate_configuration() if faster: # The faster backend always voice-clones using a reference voice # configured on the server, so no local reference audio is needed. self.tts = FasterTTSClient(voice=faster_voice) else: self.tts = QwenTTSClient( voice_mode=voice_mode, voice_clone_ref_audio=voice_clone_ref_audio, voice_clone_ref_text=voice_clone_ref_text, skip_transcription=skip_transcription, language=self.language, ) def _validate_configuration(self) -> None: """Validate configuration settings.""" if self.voice_mode not in config.VOICE_MODES: raise ValueError( f"Unknown voice mode: {self.voice_mode!r} " f"(expected one of {config.VOICE_MODES})" ) if self.voice_mode == config.VOICE_MODE_CLONE and not self.faster: if not self.voice_clone_ref_audio: raise ValueError( "Voice Clone mode requires a reference audio file. " "Use --clone to specify it." ) if not Path(self.voice_clone_ref_audio).exists(): raise ValueError( f"Reference audio file not found: {self.voice_clone_ref_audio}" ) @staticmethod def _sanitize_filename(name: str, fallback: str = "chapter") -> str: """Make a chapter title safe to use as part of a file name.""" cleaned = re.sub(r'[\\/:*?"<>|]', " ", name) cleaned = re.sub(r"\s+", " ", cleaned).strip().strip(".") return cleaned[:80] or fallback def _narrator_tag(self) -> str: """Narrator name used in output file names. Custom voice mode uses the built-in speaker's display name; voice clone mode uses the reference audio file's stem; the faster backend uses the server-side voice name. Spaces become underscores (e.g. "Uncle Fu" -> "Uncle_Fu"). """ if self.faster: narrator = self.faster_voice or config.FASTER_TTS_VOICE elif self.voice_mode == config.VOICE_MODE_CLONE: narrator = Path(self.voice_clone_ref_audio).stem else: narrator = speaker_display_name() return self._sanitize_filename(narrator, fallback="narrator").replace(" ", "_") # ------------------------------------------------------------------ # Debug dumps (--debug) # ------------------------------------------------------------------ @staticmethod def _write_debug_text(debug_dir: Path, chunk_num: int, text: str) -> None: """Write the exact text sent for a chunk to the debug folder. Called before the request so the text survives a crash mid-generation. A failed debug write must never abort a conversion. """ try: debug_dir.mkdir(parents=True, exist_ok=True) (debug_dir / f"chunk_{chunk_num:04d}.txt").write_text(text, encoding="utf-8") except OSError as exc: logger.warning("Could not write debug text for chunk %d: %s", chunk_num, exc) @staticmethod def _copy_debug_audio(debug_dir: Path, chunk_num: int, source: Path) -> Optional[Path]: """Copy a generated chunk's audio file into the debug folder. Returns the copy's path, or None when the copy failed (which never affects the conversion itself). """ try: debug_dir.mkdir(parents=True, exist_ok=True) target = debug_dir / f"chunk_{chunk_num:04d}{source.suffix or '.wav'}" shutil.copy2(source, target) return target except OSError as exc: logger.warning("Could not write debug audio for chunk %d: %s", chunk_num, exc) return None @staticmethod def _chapter_debug_dir(book_debug_dir: Optional[Path], index: int, title: str) -> Optional[Path]: """Per-chapter subfolder of a book's debug folder (None when not debugging). Chunk numbering restarts for each chapter, so chapters get their own subfolder (e.g. debug/dune_Vivian/03_The_Trial/). """ if book_debug_dir is None: return None return book_debug_dir / f"{index:02d}_{AudiobookConverter._sanitize_filename(title)}" def convert_book(self, file_path: Path, output_name: Optional[str] = None) -> bool: """Convert a single book to one or more audiobook files.""" logger.info("Converting: %s", file_path.name) start_time = time.time() try: # Start from a clean scratch folder so a previous crash can never # affect this run audio.cleanup_chunks() logger.info("Extracting text...") book = extractors.extract_book(file_path) sections = book.sections if not sections or all(not s.text.strip() for s in sections): logger.error("No text extracted") return False stem = output_name or f"{file_path.stem}_{self._narrator_tag()}" # --debug: chunk text/audio dumps land in a per-book folder debug_dir = config.DEBUG_FOLDER / stem if self.debug else None # Cover art: generated once per book. Named with the chunk_ # prefix so cleanup_chunks() removes it with the other scratch # files at the end of the book. cover_path = cover.generate_cover( book.title, config.CHUNKS_FOLDER / "chunk_cover.png") if cover_path: print(f"[INFO] Generated cover art for '{book.title}'") meta = TrackMeta(title=book.title, artist=book.author, album=book.title) # m4b is always a single file; multi-chapter books get embedded # chapter markers so listeners can skip between chapters. if self.output_format == "m4b": if len(sections) > 1: return self._convert_m4b_with_chapters(sections, stem, start_time, meta=meta, cover=cover_path, debug_dir=debug_dir) output_path = config.AUDIOBOOKS_FOLDER / f"{stem}.{self.output_format}" return self._convert_text(sections[0].text, output_path, start_time, meta=meta, cover=cover_path, debug_dir=debug_dir) if self.single_file or len(sections) == 1: text = "\n\n".join(section.text for section in sections) output_path = config.AUDIOBOOKS_FOLDER / f"{stem}.{self.output_format}" return self._convert_text(text, output_path, start_time, meta=meta, cover=cover_path, debug_dir=debug_dir) success = True for index, section in enumerate(sections, 1): chapter_name = f"{stem}_{index:02d}_{self._sanitize_filename(section.title)}" output_path = config.AUDIOBOOKS_FOLDER / f"{chapter_name}.{self.output_format}" track_meta = meta._replace( title=(section.title or "").strip() or f"Chapter {index}", track=index, total_tracks=len(sections)) success = self._convert_text( section.text, output_path, time.time(), meta=track_meta, cover=cover_path, debug_dir=self._chapter_debug_dir(debug_dir, index, section.title) ) and success return success except Exception as exc: logger.error("Conversion failed: %s", exc) logger.error(traceback.format_exc()) return False finally: # Always cleanup, even on failure or interrupt audio.cleanup_chunks() def _convert_m4b_with_chapters(self, sections, stem: str, start_time: float, meta: Optional[TrackMeta] = None, cover: Optional[Path] = None, debug_dir: Optional[Path] = None) -> bool: """Convert each chapter to audio, then assemble a single m4b with embedded chapter markers. Chapters are synthesized to lossless WAV scratch files (~170 MB per hour of audio) so the final AAC pass is the only lossy encode. When ``debug_dir`` is given, each chapter's debug dumps land in its own subfolder (chunk numbering restarts per chapter). """ chapter_files = [] titles = [] total_chapters = len(sections) for index, section in enumerate(sections, 1): chapter_path = config.CHUNKS_FOLDER / f"chapter_{index:04d}.wav" title = (section.title or "").strip() or f"Chapter {index}" print(f"\n{'=' * 50}") print(f"CHAPTER {index}/{total_chapters}: {title}") print(f"{'=' * 50}") logger.info("Converting chapter %d/%d: %s", index, total_chapters, title) if not self._convert_text(section.text, chapter_path, time.time(), speed=1.0, output_format="wav", chapter=(index, total_chapters), debug_dir=self._chapter_debug_dir(debug_dir, index, title)): logger.warning("Skipping chapter %d (%s) due to conversion failure", index, title) continue chapter_files.append(chapter_path) titles.append(title) if not chapter_files: logger.error("No chapters were successfully converted") return False output_path = config.AUDIOBOOKS_FOLDER / f"{stem}.{self.output_format}" if not audio.combine_chapters_to_m4b(chapter_files, titles, output_path, speed=self.speed, meta=meta, cover=cover): return False duration = time.time() - start_time logger.info("Conversion completed in %dm %ds: %s", int(duration // 60), int(duration % 60), output_path) print(f"[SUCCESS] Conversion completed in {int(duration // 60)}m {int(duration % 60)}s") return True def _synthesize_chunks(self, chunks: List[str], debug_dir: Optional[Path] = None) -> Dict[int, Optional[Path]]: """Synthesize chunks sequentially, preserving order and naming. Returns a mapping of chunk number to the generated audio path, with None for chunks that failed after retries. When ``debug_dir`` is given (--debug), each chunk's request text and returned audio are also dumped there, and every request/response is logged. """ total_chunks = len(chunks) print(f"\n{'=' * 50}") print(f"PROCESSING {total_chunks} CHUNKS") print(f"{'=' * 50}") results: Dict[int, Optional[Path]] = {} for chunk_num, chunk_text in enumerate(chunks, 1): if debug_dir is not None: # Written before the request so the exact text survives a # crash mid-generation; failed chunks keep their dumps. self._write_debug_text(debug_dir, chunk_num, chunk_text) logger.debug("Chunk %d/%d request text: %s", chunk_num, total_chunks, chunk_text) request_start = time.time() try: result = self.tts.process_chunk_with_retry(chunk_num, chunk_text) results[chunk_num] = result if result: if debug_dir is not None: copied = self._copy_debug_audio(debug_dir, chunk_num, Path(result)) elapsed = time.time() - request_start destination = f" -> {copied.name}" if copied else "" logger.debug("Chunk %d/%d response in %.1fs%s", chunk_num, total_chunks, elapsed, destination) print(f"[OK] Chunk {chunk_num:3d}/{total_chunks} completed") logger.info("+ Chunk %d/%d completed", chunk_num, total_chunks) else: print(f"[FAIL] Chunk {chunk_num:3d}/{total_chunks} FAILED") logger.error("- Chunk %d/%d failed", chunk_num, total_chunks) except Exception as exc: results[chunk_num] = None print(f"[ERROR] Chunk {chunk_num:3d}/{total_chunks} ERROR: {exc}") logger.error("- Chunk %d/%d error: %s", chunk_num, total_chunks, exc) successful_chunks = sum(1 for path in results.values() if path) print(f"\n{'=' * 50}") print("CHUNK PROCESSING COMPLETE") print(f"Successful: {successful_chunks}/{total_chunks}") print(f"{'=' * 50}") logger.info("Qwen processing completed: %d/%d chunks", successful_chunks, total_chunks) return results def _convert_text(self, text: str, output_path: Path, start_time: float, speed: Optional[float] = None, output_format: Optional[str] = None, chapter: Optional[Tuple[int, int]] = None, meta: Optional[TrackMeta] = None, cover: Optional[Path] = None, debug_dir: Optional[Path] = None) -> bool: """Chunk, synthesize, and assemble ``text`` into ``output_path``. When ``chapter`` (a ``(number, total)`` pair) is given, the output is an intermediate per-chapter file and progress messages are phrased accordingly instead of implying the whole book is done. ``debug_dir`` (from --debug) receives the chunks' text and audio dumps. """ if speed is None: speed = self.speed if output_format is None: output_format = self.output_format try: if not text.strip(): logger.error("No text to convert for %s", output_path.name) return False logger.info("Extracted %d characters (%d words)", len(text), len(text.split())) # Split into chunks chunks = chunking.split_into_chunks(text) total_chunks = len(chunks) if total_chunks == 0: logger.error("No chunks created") return False # Log chunk info chunk_sizes = [len(chunk.split()) for chunk in chunks] avg_chunk_size = sum(chunk_sizes) / len(chunk_sizes) logger.info("Split into %d chunks (avg %.0f words per chunk)", total_chunks, avg_chunk_size) print(f"[INFO] Processing {total_chunks} chunks via Qwen API...") results = self._synthesize_chunks(chunks, debug_dir=debug_dir) successful_chunks = sum(1 for path in results.values() if path) if successful_chunks == 0: logger.error("No chunks were successfully processed") return False if successful_chunks < total_chunks: logger.warning("Only %d/%d chunks succeeded. Proceeding with partial audiobook.", successful_chunks, total_chunks) # Combine chunks (only the successful ones) success = audio.combine_chunks(total_chunks, output_path, chunk_results=results, speed=speed, output_format=output_format, intermediate=chapter is not None, meta=meta, cover=cover) if success: duration = time.time() - start_time minutes = int(duration // 60) seconds = int(duration % 60) if chapter is not None: logger.info("Chapter %d/%d converted in %dm %ds (%d/%d chunks)", chapter[0], chapter[1], minutes, seconds, successful_chunks, total_chunks) print(f"[INFO] Chapter {chapter[0]}/{chapter[1]} converted " f"({successful_chunks}/{total_chunks} chunks)") else: logger.info("Conversion completed in %dm %ds: %s", minutes, seconds, output_path) print(f"[SUCCESS] Conversion completed in {minutes}m {seconds}s") else: logger.error("Failed to combine chunks into final audiobook") return success except Exception as exc: logger.error("Conversion failed: %s", exc) logger.error(traceback.format_exc()) return False def _print_banner(self) -> None: """Print the startup summary for the selected backend.""" print("=" * 70) print("QWEN-BASED AUDIOBOOK CONVERTER") print("=" * 70) print(f"Books folder: {config.BOOKS_FOLDER}") print(f"Output folder: {config.AUDIOBOOKS_FOLDER}") if self.faster: print(f"Faster TTS endpoint: {config.FASTER_TTS_API_URL}") print("Backend: faster (voice cloning, reference configured on server)") print(f"Voice: {self.faster_voice or config.FASTER_TTS_VOICE}") else: api_url = (config.VOICE_CLONE_API_URL if self.voice_mode == config.VOICE_MODE_CLONE else config.QWEN_API_URL) print(f"Qwen API endpoint: {api_url}") print(f"Voice mode: {self.voice_mode}") print("Model size: 1.7B (always)") if self.voice_mode == config.VOICE_MODE_CUSTOM: print(f"Speaker: {config.CUSTOM_VOICE_SPEAKER}") print(f"Language: {self.language}") elif self.voice_mode == config.VOICE_MODE_CLONE: print(f"Reference audio: {Path(self.voice_clone_ref_audio).name}") print(f"Language: {self.language}") print(f"Output format: {self.output_format}") if self.single_file and self.output_format != "m4b": print("Chapter mode: single file (--single-file)") if abs(self.speed - 1.0) >= 1e-6: print(f"Playback speed: {self.speed:g}x") if self.debug: print(f"Debug dumps (per-chunk text + raw audio): {config.DEBUG_FOLDER}") print("=" * 70) def run(self) -> bool: """Main conversion process. Returns True if all books converted.""" run_start = time.time() self._print_banner() # Check for books book_files = sorted( f for f in config.BOOKS_FOLDER.iterdir() if f.is_file() and f.suffix.lower() in config.SUPPORTED_FORMATS ) if not book_files: print(f"[INFO] No supported files found in {config.BOOKS_FOLDER}") print(f"Supported formats: {', '.join(config.SUPPORTED_FORMATS)}") # Create sample file sample_file = config.BOOKS_FOLDER / "sample.txt" sample_file.write_text( "This is a sample audiobook for testing the Qwen-based converter. " "The system will send this text to the Qwen API for voice generation. " "You can replace this file with your own books to convert.", encoding="utf-8", ) print(f"[INFO] Created sample file: {sample_file}") return True print(f"[INFO] Found {len(book_files)} books to convert") # Avoid output collisions when two books share a stem (e.g. dune.txt + dune.epub). stem_counts: Dict[str, int] = Counter(book_file.stem for book_file in book_files) # Ask every overwrite question up front, before any conversion # starts, so the rest of the run is unattended. planned: List[Tuple[Path, str]] = [] for book_file in book_files: output_name = book_file.stem if stem_counts[book_file.stem] > 1: output_name = f"{book_file.stem}_{book_file.suffix.lstrip('.')}" output_name = f"{output_name}_{self._narrator_tag()}" existing = find_existing_outputs(output_name, self.output_format) if existing and not prompt_overwrite(existing, output_name): print(f"[INFO] Skipping {book_file.name} (existing output kept)") continue planned.append((book_file, output_name)) if not planned: print("[INFO] Nothing to convert (all books skipped)") return True print(f"[INFO] Converting {len(planned)} of {len(book_files)} book(s)") # Convert each book results = {} for book_file, output_name in planned: try: success = self.convert_book(book_file, output_name=output_name) results[book_file.name] = success except KeyboardInterrupt: print("\n[WARNING] Conversion interrupted by user") results[book_file.name] = False break except Exception as exc: logger.error("Unexpected error: %s", exc) results[book_file.name] = False # Print summary successful = sum(results.values()) total = len(results) print("\n" + "=" * 70) print("CONVERSION SUMMARY") print("=" * 70) print(f"Total: {total} | Success: {successful} | Failed: {total - successful}") print("=" * 70) for filename, success in results.items(): status = "[OK]" if success else "[FAIL]" print(f"{status} {filename}") if successful > 0: print(f"\n[INFO] Audiobooks saved to: {config.AUDIOBOOKS_FOLDER}/") elapsed = int(time.time() - run_start) hours, remainder = divmod(elapsed, 3600) minutes, seconds = divmod(remainder, 60) if hours: duration = f"{hours}h {minutes}m {seconds}s" elif minutes: duration = f"{minutes}m {seconds}s" else: duration = f"{seconds}s" print(f"\n[INFO] Generation completed in {duration}") logger.info("Generation completed in %s", duration) return total > 0 and successful == total