"""Orchestrates book-to-audiobook conversion.""" import logging import re import sys import time import traceback from datetime import datetime from pathlib import Path from typing import Dict, Optional from . import audio, chunking, config, extractors from .tts import QwenTTSClient logger = logging.getLogger(__name__) def setup_logging() -> None: """Configure logging to both a dated file and the console.""" config.LOGS_FOLDER.mkdir(parents=True, exist_ok=True) logging.basicConfig( level=logging.INFO, format="%(asctime)s - %(levelname)s - %(message)s", handlers=[ logging.FileHandler( config.LOGS_FOLDER / f"audiobook_{datetime.now():%Y%m%d}.log", encoding="utf-8", ), logging.StreamHandler(sys.stdout), ], ) def setup_directories() -> None: """Create necessary directories.""" for directory in (config.BOOKS_FOLDER, config.AUDIOBOOKS_FOLDER, config.CHUNKS_FOLDER, config.LOGS_FOLDER): Path(directory).mkdir(parents=True, exist_ok=True) class AudiobookConverter: """Audiobook converter using the Qwen TTS API.""" def __init__(self, voice_mode: str = "custom_voice", voice_clone_ref_audio: Optional[str] = None, voice_clone_ref_text: Optional[str] = None, skip_transcription: bool = False, speed: float = 1.0, single_file: bool = False, output_format: str = "mp3"): if speed <= 0: raise ValueError(f"Speed must be a positive number, got {speed}") if output_format not in config.AUDIO_FORMATS: raise ValueError(f"Unsupported output format: {output_format}") self.voice_mode = voice_mode self.voice_clone_ref_audio = voice_clone_ref_audio self.speed = speed self.single_file = single_file self.output_format = output_format self._validate_configuration() self.tts = QwenTTSClient( voice_mode=voice_mode, voice_clone_ref_audio=voice_clone_ref_audio, voice_clone_ref_text=voice_clone_ref_text, skip_transcription=skip_transcription, ) def _validate_configuration(self) -> None: """Validate configuration settings.""" if self.voice_mode == "voice_clone": if not self.voice_clone_ref_audio: raise ValueError( "Voice Clone mode requires a reference audio file. " "Use --voice-sample to specify it." ) if not Path(self.voice_clone_ref_audio).exists(): raise ValueError( f"Reference audio file not found: {self.voice_clone_ref_audio}" ) @staticmethod def _sanitize_filename(name: str) -> str: """Make a chapter title safe to use as part of a file name.""" cleaned = re.sub(r'[\\/:*?"<>|]', " ", name) cleaned = re.sub(r"\s+", " ", cleaned).strip().strip(".") return cleaned[:80] or "chapter" def convert_book(self, file_path: Path, output_name: Optional[str] = None) -> bool: """Convert a single book to one or more audiobook files.""" logger.info("Converting: %s", file_path.name) start_time = time.time() try: # Start from a clean scratch folder so a previous crash can never # affect this run audio.cleanup_chunks() logger.info("Extracting text...") sections = extractors.extract_sections(file_path) if not sections or all(not s.text.strip() for s in sections): logger.error("No text extracted") return False stem = output_name or file_path.stem # m4b is always a single file; multi-chapter books get embedded # chapter markers so listeners can skip between chapters. if self.output_format == "m4b": if len(sections) > 1: return self._convert_m4b_with_chapters(sections, stem, start_time) output_path = config.AUDIOBOOKS_FOLDER / f"{stem}.{self.output_format}" return self._convert_text(sections[0].text, output_path, start_time) if self.single_file or len(sections) == 1: text = "\n\n".join(section.text for section in sections) output_path = config.AUDIOBOOKS_FOLDER / f"{stem}.{self.output_format}" return self._convert_text(text, output_path, start_time) success = True for index, section in enumerate(sections, 1): chapter_name = f"{stem}_{index:02d}_{self._sanitize_filename(section.title)}" output_path = config.AUDIOBOOKS_FOLDER / f"{chapter_name}.{self.output_format}" success = self._convert_text(section.text, output_path, start_time) and success return success except Exception as exc: logger.error("Conversion failed: %s", exc) logger.error(traceback.format_exc()) return False finally: # Always cleanup, even on failure or interrupt audio.cleanup_chunks() def _convert_m4b_with_chapters(self, sections, stem: str, start_time: float) -> bool: """Convert each chapter to audio, then assemble a single m4b with embedded chapter markers. Chapters are synthesized to lossless WAV scratch files (~170 MB per hour of audio) so the final AAC pass is the only lossy encode. """ chapter_files = [] titles = [] for index, section in enumerate(sections, 1): chapter_path = config.CHUNKS_FOLDER / f"chapter_{index:04d}.wav" if not self._convert_text(section.text, chapter_path, start_time, speed=1.0, output_format="wav"): logger.warning("Skipping chapter %d (%s) due to conversion failure", index, section.title) continue chapter_files.append(chapter_path) titles.append(section.title or f"Chapter {index}") if not chapter_files: logger.error("No chapters were successfully converted") return False output_path = config.AUDIOBOOKS_FOLDER / f"{stem}.{self.output_format}" return audio.combine_chapters_to_m4b(chapter_files, titles, output_path, speed=self.speed) def _convert_text(self, text: str, output_path: Path, start_time: float, speed: Optional[float] = None, output_format: Optional[str] = None) -> bool: """Chunk, synthesize, and assemble ``text`` into ``output_path``.""" if speed is None: speed = self.speed if output_format is None: output_format = self.output_format try: if not text.strip(): logger.error("No text to convert for %s", output_path.name) return False logger.info("Extracted %d characters (%d words)", len(text), len(text.split())) # Split into chunks chunks = chunking.split_into_chunks(text) total_chunks = len(chunks) if total_chunks == 0: logger.error("No chunks created") return False # Log chunk info chunk_sizes = [len(chunk.split()) for chunk in chunks] avg_chunk_size = sum(chunk_sizes) / len(chunk_sizes) logger.info("Split into %d chunks (avg %.0f words per chunk)", total_chunks, avg_chunk_size) print(f"[INFO] Processing {total_chunks} chunks via Qwen API...") print(f"[INFO] Estimated time: ~{total_chunks * 4} minutes (4 min per chunk)") print(f"\n{'=' * 50}") print(f"PROCESSING {total_chunks} CHUNKS") print(f"{'=' * 50}") # Process chunks sequentially to ensure correct order and naming: # chunks are named 1, 2, 3, 4... in order. results = {} # chunk_num -> success (bool) for chunk_num, chunk_text in enumerate(chunks, 1): try: result = self.tts.process_chunk_with_retry(chunk_num, chunk_text) results[chunk_num] = result if result: print(f"[OK] Chunk {chunk_num:3d}/{total_chunks} completed") logger.info("+ Chunk %d/%d completed", chunk_num, total_chunks) else: print(f"[FAIL] Chunk {chunk_num:3d}/{total_chunks} FAILED") logger.error("- Chunk %d/%d failed", chunk_num, total_chunks) except Exception as exc: results[chunk_num] = False print(f"[ERROR] Chunk {chunk_num:3d}/{total_chunks} ERROR: {exc}") logger.error("- Chunk %d/%d error: %s", chunk_num, total_chunks, exc) successful_chunks = sum(1 for v in results.values() if v) print(f"\n{'=' * 50}") print("CHUNK PROCESSING COMPLETE") print(f"Successful: {successful_chunks}/{total_chunks}") print(f"{'=' * 50}") logger.info("Qwen processing completed: %d/%d chunks", successful_chunks, total_chunks) if successful_chunks == 0: logger.error("No chunks were successfully processed") return False if successful_chunks < total_chunks: logger.warning("Only %d/%d chunks succeeded. Proceeding with partial audiobook.", successful_chunks, total_chunks) # Combine chunks (only the successful ones) success = audio.combine_chunks(total_chunks, output_path, results, speed=speed, output_format=output_format) if success: duration = time.time() - start_time minutes = int(duration // 60) seconds = int(duration % 60) logger.info("Conversion completed in %dm %ds: %s", minutes, seconds, output_path) print(f"[SUCCESS] Conversion completed in {minutes}m {seconds}s") else: logger.error("Failed to combine chunks into final audiobook") return success except Exception as exc: logger.error("Conversion failed: %s", exc) logger.error(traceback.format_exc()) return False def run(self) -> bool: """Main conversion process. Returns True if all books converted.""" api_url = config.VOICE_CLONE_API_URL if self.voice_mode == "voice_clone" else config.QWEN_API_URL print("=" * 70) print("QWEN-BASED AUDIOBOOK CONVERTER") print("=" * 70) print(f"Books folder: {config.BOOKS_FOLDER}") print(f"Output folder: {config.AUDIOBOOKS_FOLDER}") print(f"Qwen API endpoint: {api_url}") print(f"Voice mode: {self.voice_mode}") print("Model size: 1.7B (always)") if self.voice_mode == "custom_voice": print(f"Speaker: {config.CUSTOM_VOICE_SPEAKER}") print(f"Language: {config.CUSTOM_VOICE_LANGUAGE}") elif self.voice_mode == "voice_clone": print(f"Reference audio: {Path(self.voice_clone_ref_audio).name}") print(f"Language: {config.VOICE_CLONE_LANGUAGE}") print(f"Output format: {self.output_format}") if self.single_file and self.output_format == "mp3": print("Chapter mode: single file (--single-file)") if abs(self.speed - 1.0) >= 1e-6: print(f"Playback speed: {self.speed:g}x") print("=" * 70) # Check for books book_files = sorted( f for f in config.BOOKS_FOLDER.iterdir() if f.is_file() and f.suffix.lower() in config.SUPPORTED_FORMATS ) if not book_files: print(f"[INFO] No supported files found in {config.BOOKS_FOLDER}") print(f"Supported formats: {', '.join(config.SUPPORTED_FORMATS)}") # Create sample file sample_file = config.BOOKS_FOLDER / "sample.txt" sample_file.write_text( "This is a sample audiobook for testing the Qwen-based converter. " "The system will send this text to the Qwen API for voice generation. " "You can replace this file with your own books to convert.", encoding="utf-8", ) print(f"[INFO] Created sample file: {sample_file}") return True print(f"[INFO] Found {len(book_files)} books to convert") # Avoid output collisions when two books share a stem (e.g. dune.txt + dune.epub). stem_counts: Dict[str, int] = {} for book_file in book_files: stem_counts[book_file.stem] = stem_counts.get(book_file.stem, 0) + 1 # Convert each book results = {} for book_file in book_files: output_name = book_file.stem if stem_counts[book_file.stem] > 1: output_name = f"{book_file.stem}_{book_file.suffix.lstrip('.')}" try: success = self.convert_book(book_file, output_name=output_name) results[book_file.name] = success except KeyboardInterrupt: print("\n[WARNING] Conversion interrupted by user") break except Exception as exc: logger.error("Unexpected error: %s", exc) results[book_file.name] = False # Print summary successful = sum(results.values()) total = len(results) print("\n" + "=" * 70) print("CONVERSION SUMMARY") print("=" * 70) print(f"Total: {total} | Success: {successful} | Failed: {total - successful}") print("=" * 70) for filename, success in results.items(): status = "[OK]" if success else "[FAIL]" print(f"{status} {filename}") if successful > 0: print(f"\n[INFO] Audiobooks saved to: {config.AUDIOBOOKS_FOLDER}/") return total > 0 and successful == total