diff options
| author | historia <historiavg@proton.me> | 2026-08-24 02:59:26 -0400 |
|---|---|---|
| committer | historia <historiavg@proton.me> | 2026-08-24 02:59:26 -0400 |
| commit | f00249db9d1ea051d29aa1bcca869fc4b88e83eb (patch) | |
| tree | a75f076fac1b63e0b4bf2eb8f54affbcc681a891 /converter/audio.py | |
| parent | 9dd4f9595be3b1d76a3a07dc3eca90cfaf8a3f97 (diff) | |
| download | tts-audiobook-generator-f00249db9d1ea051d29aa1bcca869fc4b88e83eb.tar.gz | |
refactor: add app directory, dir structure change
Diffstat (limited to 'converter/audio.py')
| -rw-r--r-- | converter/audio.py | 616 |
1 files changed, 0 insertions, 616 deletions
diff --git a/converter/audio.py b/converter/audio.py deleted file mode 100644 index d1dc28b..0000000 --- a/converter/audio.py +++ /dev/null @@ -1,616 +0,0 @@ -"""Audio assembly: combining chunks, speed adjustment, cleanup.""" - -import logging -import re -import shutil -import subprocess -import traceback -import wave -from pathlib import Path -from typing import Dict, List, NamedTuple, Optional, Tuple - -from . import config - -logger = logging.getLogger(__name__) - -CHUNKS_FOLDER = Path(__file__).resolve().parent.parent / "chunks" - - -def atempo_filters(speed: float) -> str: - """Return a comma-joined ffmpeg ``atempo`` filter chain for ``speed``. - - ``atempo`` accepts 0.5..2.0 per filter; values outside that range are - handled by chaining multiple filters. Returns "" when ``speed`` is 1.0. - """ - if speed <= 0: - raise ValueError(f"Speed must be a positive number, got {speed}") - if abs(speed - 1.0) < 1e-6: - return "" - remaining = float(speed) - chain = [] - while remaining > 2.0: - chain.append("atempo=2.0") - remaining /= 2.0 - while remaining < 0.5: - chain.append("atempo=0.5") - remaining /= 0.5 - chain.append(f"atempo={remaining:g}") - return ",".join(chain) - - -def speed_export_params(speed: float) -> List[str]: - """Return ffmpeg filter args for pitch-preserving speed adjustment.""" - filters = atempo_filters(speed) - if not filters: - return [] - return ["-filter:a", filters] - - -def _concat_escape(path: str) -> str: - """Escape a path for use inside single quotes in an ffmpeg concat list.""" - return path.replace("'", "'\\''") - - -def _concat_wav_files(sources: List[Path], destination: Path) -> bool: - """Concatenate WAV files with matching parameters using the wave module. - - Returns False (touching nothing) when any input is not a readable WAV - or the parameters differ, so the caller can fall back to ffmpeg. - """ - opened = [] - try: - parameters = None - for source in sources: - wav_file = wave.open(str(source), "rb") - opened.append(wav_file) - current = (wav_file.getnchannels(), wav_file.getsampwidth(), - wav_file.getframerate()) - if parameters is None: - parameters = current - elif current != parameters: - return False - if parameters is None or min(parameters) < 1: - return False - with wave.open(str(destination), "wb") as output: - output.setnchannels(parameters[0]) - output.setsampwidth(parameters[1]) - output.setframerate(parameters[2]) - for wav_file in opened: - output.writeframes(wav_file.readframes(wav_file.getnframes())) - return True - except (wave.Error, EOFError, OSError): - return False - finally: - for wav_file in opened: - try: - wav_file.close() - except Exception: - pass - - -def _concat_with_ffmpeg(sources: List[Path], destination: Path) -> None: - """Concatenate audio files with ffmpeg's concat demuxer, re-encoding to - 16-bit PCM WAV (handles inputs the wave module cannot).""" - if shutil.which("ffmpeg") is None: - raise RuntimeError( - "ffmpeg is required to concatenate audio parts in non-WAV formats " - "(install ffmpeg and try again)" - ) - list_path = destination.with_name(destination.stem + "_parts.txt") - try: - with open(list_path, "w", encoding="utf-8") as list_file: - for source in sources: - list_file.write(f"file '{_concat_escape(str(source))}'\n") - command = [ - "ffmpeg", "-y", "-hide_banner", "-loglevel", "error", - "-f", "concat", "-safe", "0", "-i", str(list_path), - "-c:a", "pcm_s16le", str(destination), - ] - proc = subprocess.run(command, capture_output=True, text=True) - if proc.returncode != 0: - raise RuntimeError( - f"ffmpeg failed to concatenate audio parts: {proc.stderr[-500:]}") - finally: - try: - list_path.unlink(missing_ok=True) - except OSError: - pass - - -def concat_audio_files(sources: List[Path], destination: Path) -> None: - """Concatenate audio files into one file at ``destination``. - - Joins the audio returned by several TTS sub-requests for a single - chunk. Uses the stdlib wave module when every input is a WAV with - matching parameters (lossless, no external tools); otherwise falls - back to ffmpeg's concat demuxer with re-encoding. - """ - if not sources: - raise ValueError("No audio files to concatenate") - if _concat_wav_files(sources, destination): - return - _concat_with_ffmpeg(sources, destination) - - -def _encode_args(output_format: str) -> List[str]: - """Return ffmpeg output codec/bitrate args for the requested container.""" - if output_format == "m4b": - return ["-c:a", "aac", "-b:a", config.AUDIO_BITRATE] - if output_format == "ogg": - return ["-c:a", "libvorbis", "-b:a", config.AUDIO_BITRATE] - if output_format == "flac": - return ["-c:a", "flac"] - if output_format == "wav": - # Lossless intermediate for per-chapter scratch audio; avoids - # generational loss when the final m4b re-encodes to AAC. - return ["-c:a", "pcm_s16le"] - return ["-b:a", config.AUDIO_BITRATE] - - -class TrackMeta(NamedTuple): - """Tags embedded into a finished audiobook file.""" - - title: str - artist: str = "" - album: str = "" - track: Optional[int] = None - total_tracks: Optional[int] = None - - -def _tag_args(meta: TrackMeta, output_format: str) -> List[str]: - """Return -metadata args plus format-specific tagging flags.""" - args = ["-metadata", f"title={meta.title}"] - if meta.artist: - args += ["-metadata", f"artist={meta.artist}"] - if meta.album: - args += ["-metadata", f"album={meta.album}"] - if meta.track and meta.total_tracks: - args += ["-metadata", f"track={meta.track}/{meta.total_tracks}"] - if output_format == "mp3": - # ID3v2.3 is what essentially every player reads; ffmpeg's default - # (v2.4) still confuses some of them. - args += ["-id3v2_version", "3"] - return args - - -# Formats whose container has no reliable embedded-picture support. -_NO_COVER_FORMATS = ("ogg", "wav") - - -def _cover_args(output_format: str, cover_input_index: int) -> List[str]: - """Return per-output args attaching a cover image as an embedded picture. - - The cover must already be added as an ffmpeg input; ``cover_input_index`` - is that input's position on the command line. mp3/flac keep the PNG - stream as-is; m4b re-encodes to JPEG, which audiobook players expect. - """ - if output_format in _NO_COVER_FORMATS: - return [] - codec = "mjpeg" if output_format == "m4b" else "copy" - args = ["-map", f"{cover_input_index}:v", "-c:v", codec, - "-disposition:v", "attached_pic", - "-metadata:s:v", "title=Album cover"] - if output_format == "m4b": - args += ["-q:v", "3"] - return args - - -_brand_supported: Optional[bool] = None - - -def _detect_brand_support() -> bool: - """Check whether the local ffmpeg muxer accepts the ``-brand`` option.""" - if shutil.which("ffmpeg") is None: - return False - try: - proc = subprocess.run( - ["ffmpeg", "-hide_banner", "-h", "muxer=ipod"], - capture_output=True, text=True, timeout=15, - ) - except (OSError, subprocess.TimeoutExpired): - return False - return "-brand" in proc.stdout - - -def _m4b_container_args() -> List[str]: - """Return per-output container flags for m4b files. - - ``+faststart`` moves the ``moov`` index to the front of the file so - streaming players (and naive linear readers like web players) can index - it; without it they may misreport the duration or refuse the file. - The ``M4B `` major brand identifies the file as an audiobook to - players that sniff brands instead of trusting the extension (the ffmpeg - default brand for .m4b is ``M4A ``). - """ - global _brand_supported - if _brand_supported is None: - _brand_supported = _detect_brand_support() - args = ["-movflags", "+faststart"] - if _brand_supported: - args += ["-brand", "M4B "] - return args - - -def build_concat_command(concat_list: Path, output_path: Path, output_format: str, - speed: float = 1.0, speed_path: Optional[Path] = None, - meta: Optional[TrackMeta] = None, - cover: Optional[Path] = None) -> List[str]: - """Build the ffmpeg command that concatenates chunk audio into a book file.""" - encode = _encode_args(output_format) - container = _m4b_container_args() if output_format == "m4b" else [] - inputs = ["-f", "concat", "-safe", "0", "-i", str(concat_list)] - cover_index = None - if cover is not None and output_format not in _NO_COVER_FORMATS: - inputs += ["-i", str(cover)] - cover_index = 1 - cover_block = (_cover_args(output_format, cover_index) - if cover_index is not None else []) - tags = _tag_args(meta, output_format) if meta else [] - filters = atempo_filters(speed) - if filters: - if speed_path is None: - raise ValueError("speed_path is required when speed is not 1.0") - return [ - "ffmpeg", "-y", *inputs, "-filter_complex", - f"[0:a]split=2[base][spd];[spd]{filters}[spdout]", - "-map", "[base]", *encode, *tags, *cover_block, *container, str(output_path), - "-map", "[spdout]", *encode, *tags, *cover_block, *container, str(speed_path), - ] - output_maps = ["-map", "0:a"] if cover_block else [] - return [ - "ffmpeg", "-y", *inputs, *output_maps, - *encode, *tags, *cover_block, *container, str(output_path), - ] - - -def build_m4b_chapters_command(concat_list: Path, metadata_file: Path, output_path: Path, - speed: float = 1.0, speed_path: Optional[Path] = None, - speed_metadata_file: Optional[Path] = None, - meta: Optional[TrackMeta] = None, - cover: Optional[Path] = None) -> List[str]: - """Build the ffmpeg command that assembles chapter audio into one m4b. - - Chapter metadata inputs are bound to their outputs with explicit - ``-map_chapters`` so the normal-speed and speed-adjusted copies each get - their own (correctly scaled) chapter markers. - """ - encode = _encode_args("m4b") - container = _m4b_container_args() - tags = _tag_args(meta, "m4b") if meta else [] - inputs = ["-f", "concat", "-safe", "0", "-i", str(concat_list), - "-i", str(metadata_file)] - input_count = 2 # concat audio + ffmetadata - if speed_path is not None: - inputs += ["-i", str(speed_metadata_file)] - input_count += 1 - cover_block: List[str] = [] - if cover is not None: - inputs += ["-i", str(cover)] - cover_block = _cover_args("m4b", input_count) - if speed_path is not None: - return [ - "ffmpeg", "-y", *inputs, "-filter_complex", - f"[0:a]split=2[base][spd];[spd]{atempo_filters(speed)}[spdout]", - "-map", "[base]", *cover_block, "-map_metadata", "1", "-map_chapters", "1", - *encode, *tags, *container, str(output_path), - "-map", "[spdout]", *cover_block, "-map_metadata", "2", "-map_chapters", "2", - *encode, *tags, *container, str(speed_path), - ] - return [ - "ffmpeg", "-y", *inputs, - "-map", "0:a", *cover_block, "-map_metadata", "1", "-map_chapters", "1", - *encode, *tags, *container, str(output_path), - ] - - -_DURATION_WARN_TOLERANCE = 0.05 # warn when output duration drifts >5% -_DURATION_FAIL_TOLERANCE = 0.25 # fail when output duration drifts >25% - - -def verify_output_duration(path: Path, expected_ms: int) -> bool: - """Sanity-check an assembled file's duration against the expected total. - - Catches corrupt assembly (truncated concat, bogus container metadata) - before the file reaches audiobook players. Duration drift beyond the - warn tolerance is logged; drift beyond the fail tolerance is an error - and the output is treated as broken. Returns True when unverifiable. - """ - if expected_ms <= 0: - return True - actual_ms = probe_duration_ms(path) - if actual_ms <= 0: - logger.warning("Could not verify duration of %s (ffprobe failed)", path) - return True - drift = abs(actual_ms - expected_ms) / expected_ms - if drift > _DURATION_FAIL_TOLERANCE: - logger.error( - "Duration mismatch for %s: expected ~%.1fs, got %.1fs (%.0f%% off); output is likely corrupt", - path.name, expected_ms / 1000.0, actual_ms / 1000.0, drift * 100.0, - ) - return False - if drift > _DURATION_WARN_TOLERANCE: - logger.warning( - "Duration drift for %s: expected ~%.1fs, got %.1fs (%.0f%% off)", - path.name, expected_ms / 1000.0, actual_ms / 1000.0, drift * 100.0, - ) - return True - - -def _collect_chunk_files(total_chunks: int, - chunk_results: Dict[int, Optional[Path]] - ) -> Tuple[List[Path], List[int]]: - """Resolve chunk audio files in book order. - - ``chunk_results`` maps chunk number -> path written (or None for a failed - chunk); recorded paths are used exactly as-is so stale files from a - previous chapter can never leak in. - """ - chunk_files: List[Path] = [] - missing: List[int] = [] - for i in range(1, total_chunks + 1): - recorded = chunk_results.get(i) - if recorded is not None and Path(recorded).exists(): - chunk_files.append(Path(recorded)) - else: - missing.append(i) - return chunk_files, missing - - -def combine_chunks(total_chunks: int, output_path: Path, - chunk_results: Dict[int, Optional[Path]], - speed: float = 1.0, output_format: str = config.AUDIO_FORMAT, - intermediate: bool = False, - meta: Optional[TrackMeta] = None, - cover: Optional[Path] = None) -> bool: - """Combine audio chunks into the final audiobook using ffmpeg's concat demuxer. - - ``chunk_results`` maps chunk numbers to the audio file each chunk produced - (None for failed chunks); failed and missing chunks are skipped. When - ``speed`` differs from 1.0, an additional speed-adjusted copy is written - next to the normal-speed file. ``meta``/``cover`` embed tags and cover - art into the output (skipped for intermediate chapter scratch audio). - Chunks are streamed by ffmpeg, so the whole book is never held in - memory. Set ``intermediate`` for scratch chapter audio on the way to a - larger output (e.g. a chaptered m4b) so save messages don't present it - as the final audiobook. - """ - if shutil.which("ffmpeg") is None or shutil.which("ffprobe") is None: - logger.error("ffmpeg and ffprobe are required to combine audio chunks (install ffmpeg)") - return False - - chunk_files, missing_chunks = _collect_chunk_files(total_chunks, chunk_results) - - if not chunk_files: - logger.error("No valid chunks found") - return False - - if missing_chunks: - logger.warning("Missing chunks: %s", missing_chunks) - - concat_list = CHUNKS_FOLDER / "_concat_list.txt" - try: - with open(concat_list, "w", encoding="utf-8") as list_file: - for chunk_file in chunk_files: - list_file.write(f"file '{_concat_escape(str(chunk_file))}'\n") - - speed_path = None - if atempo_filters(speed): - speed_path = output_path.with_name(f"{output_path.stem}_{speed:g}{output_path.suffix}") - cmd = build_concat_command(concat_list, output_path, output_format, - speed=speed, speed_path=speed_path, - meta=None if intermediate else meta, - cover=None if intermediate else cover) - - proc = subprocess.run(cmd, capture_output=True, text=True) - if proc.returncode != 0: - logger.error("ffmpeg failed: %s", proc.stderr[-2000:]) - return False - - # Verify the assembled duration against the sum of chunk durations - # so corrupt output is caught before it reaches audiobook players. - # A failed probe returns 0, which would deflate the expected total - # and falsely fail the check, so unverifiable sums skip it. - durations = [probe_duration_ms(chunk_file) for chunk_file in chunk_files] - if any(duration <= 0 for duration in durations): - logger.warning("Could not probe every chunk duration; skipping duration verification") - duration_ok = True - else: - expected_ms = sum(durations) - duration_ok = verify_output_duration(output_path, expected_ms) - if speed_path is not None: - duration_ok = verify_output_duration(speed_path, int(expected_ms / speed)) and duration_ok - if not duration_ok: - return False - - if intermediate: - logger.info("Chapter audio saved (intermediate): %s (%d/%d chunks)", - output_path, len(chunk_files), total_chunks) - if len(chunk_files) == 1 and total_chunks == 1: - print(f"[INFO] Saved chapter audio (intermediate): " - f"{output_path.name}") - else: - print(f"[INFO] Saved chapter audio (intermediate): {output_path.name} " - f"({len(chunk_files)}/{total_chunks} chunks)") - else: - logger.info("Audiobook saved: %s (%d/%d chunks)", output_path, len(chunk_files), total_chunks) - if len(chunk_files) == 1 and total_chunks == 1: - print(f"[INFO] Saved audiobook: {output_path.name}") - else: - print(f"[INFO] Saved audiobook: {output_path.name} " - f"({len(chunk_files)}/{total_chunks} chunks)") - - if speed_path is not None: - logger.info("Saved speed-adjusted audiobook (%gx): %s", speed, speed_path) - print(f"[INFO] Saved speed-adjusted audiobook: {speed_path.name} ({speed:g}x)") - - return True - - except FileNotFoundError: - logger.error("ffmpeg not found on PATH (install ffmpeg and try again)") - return False - except Exception as exc: - logger.error("Failed to combine chunks: %s", exc) - logger.error(traceback.format_exc()) - return False - finally: - try: - concat_list.unlink(missing_ok=True) - except OSError: - pass - - -def cleanup_chunks() -> None: - """Remove temporary chunk and chapter files from the scratch folder.""" - try: - chunk_count = 0 - for pattern in ("chunk_*", "chapter_*"): - for chunk_file in CHUNKS_FOLDER.glob(pattern): - try: - if chunk_file.is_file(): - chunk_file.unlink() - chunk_count += 1 - except Exception as exc: - logger.warning("Failed to delete %s: %s", chunk_file, exc) - - if chunk_count > 0: - logger.info("Cleaned up %d chunk files", chunk_count) - print(f"[INFO] Cleaned up {chunk_count} chunk files") - except Exception as exc: - logger.warning("Cleanup failed: %s", exc) - - -def probe_duration_ms(path: Path) -> int: - """Return audio duration in milliseconds using ffprobe.""" - try: - result = subprocess.run( - ["ffprobe", "-v", "error", "-show_entries", "format=duration", - "-of", "default=noprint_wrappers=1:nokey=1", str(path)], - capture_output=True, text=True, timeout=30, - ) - except subprocess.TimeoutExpired: - logger.warning("ffprobe timed out for %s", path) - return 0 - if result.returncode != 0: - logger.warning("ffprobe failed for %s: %s", path, result.stderr[-200:]) - return 0 - try: - return max(0, int(round(float(result.stdout.strip()) * 1000.0))) - except ValueError: - logger.warning("Could not parse ffprobe duration for %s", path) - return 0 - - -def _escape_ffmetadata_value(value: str) -> str: - """Escape a metadata value for ffmpeg's FFMETADATA format. - - Backslash and the structural characters ``=``, ``;`` and ``#`` must be - backslash-escaped; line breaks would corrupt the file and are collapsed - to spaces. - """ - value = value.replace("\\", "\\\\") - value = re.sub(r"[\r\n]+", " ", value) - return re.sub(r"[=;#]", r"\\\g<0>", value) - - -def build_ffmetadata(chapters: List[tuple], path: Path) -> None: - """Write an ffmpeg FFMETADATA file with ``[CHAPTER]`` entries. - - ``chapters`` is a list of ``(start_ms, end_ms, title)`` tuples. - """ - with open(path, "w", encoding="utf-8") as meta_file: - meta_file.write(";FFMETADATA1\n") - for start_ms, end_ms, title in chapters: - meta_file.write("[CHAPTER]\n") - meta_file.write("TIMEBASE=1/1000\n") - meta_file.write(f"START={int(start_ms)}\n") - meta_file.write(f"END={int(end_ms)}\n") - meta_file.write(f"title={_escape_ffmetadata_value(title)}\n") - - -def combine_chapters_to_m4b(chapter_files: List[Path], titles: List[str], - output_path: Path, speed: float = 1.0, - meta: Optional[TrackMeta] = None, - cover: Optional[Path] = None) -> bool: - """Concatenate per-chapter audio into a single m4b with embedded chapter markers. - - Chapter start/end times are derived from each chapter file's duration and - written as ffmpeg chapter metadata. ``meta``/``cover`` embed tags and - cover art. When ``speed`` differs from 1.0, a speed-adjusted copy (with - rescaled chapter markers) is written alongside the normal-speed file. - """ - if shutil.which("ffmpeg") is None or shutil.which("ffprobe") is None: - logger.error("ffmpeg and ffprobe are required to build an m4b with chapters") - return False - - if not chapter_files: - logger.error("No chapter files provided") - return False - - concat_list = CHUNKS_FOLDER / "_concat_list.txt" - metadata_file = CHUNKS_FOLDER / "_chapters.txt" - speed_metadata_file = CHUNKS_FOLDER / "_chapters_speed.txt" - try: - chapters = [] - start_ms = 0 - with open(concat_list, "w", encoding="utf-8") as list_file: - for chapter_file, title in zip(chapter_files, titles): - list_file.write(f"file '{_concat_escape(str(chapter_file))}'\n") - duration_ms = probe_duration_ms(chapter_file) - end_ms = start_ms + duration_ms - chapters.append((start_ms, end_ms, title or "Chapter")) - start_ms = end_ms - - build_ffmetadata(chapters, metadata_file) - - speed_path = None - if atempo_filters(speed): - speed_path = output_path.with_name(f"{output_path.stem}_{speed:g}{output_path.suffix}") - scaled = [(int(s / speed), int(e / speed), t) for s, e, t in chapters] - build_ffmetadata(scaled, speed_metadata_file) - cmd = build_m4b_chapters_command(concat_list, metadata_file, output_path, - speed=speed, speed_path=speed_path, - speed_metadata_file=speed_metadata_file - if speed_path is not None else None, - meta=meta, cover=cover) - - proc = subprocess.run(cmd, capture_output=True, text=True) - if proc.returncode != 0: - logger.error("ffmpeg failed: %s", proc.stderr[-2000:]) - return False - - # The last chapter's end time is the expected total duration; skip - # verification when any chapter duration probe failed (returned 0) - # so an unprobed chapter can't falsely fail the whole output. - expected_ms = chapters[-1][1] - if any(end_ms - start_ms <= 0 for start_ms, end_ms, _ in chapters): - logger.warning("Could not probe every chapter duration; skipping duration verification") - duration_ok = True - else: - duration_ok = verify_output_duration(output_path, expected_ms) - if speed_path is not None: - duration_ok = verify_output_duration(speed_path, int(expected_ms / speed)) and duration_ok - if not duration_ok: - return False - - logger.info("Audiobook saved: %s (%d chapters)", output_path, len(chapter_files)) - print(f"[INFO] Saved audiobook: {output_path.name} ({len(chapter_files)} chapters)") - - if speed_path is not None: - logger.info("Saved speed-adjusted audiobook (%gx): %s", speed, speed_path) - print(f"[INFO] Saved speed-adjusted audiobook: {speed_path.name} ({speed:g}x)") - - return True - - except FileNotFoundError: - logger.error("ffmpeg/ffprobe not found on PATH (install ffmpeg and try again)") - return False - except Exception as exc: - logger.error("Failed to combine chapters: %s", exc) - logger.error(traceback.format_exc()) - return False - finally: - for scratch in (concat_list, metadata_file, speed_metadata_file): - try: - scratch.unlink(missing_ok=True) - except OSError: - pass |
