aboutsummaryrefslogtreecommitdiff
path: root/converter/config.py
diff options
context:
space:
mode:
authorhistoria <historiavg@proton.me>2026-08-19 03:55:13 -0400
committerhistoria <historiavg@proton.me>2026-08-19 03:55:13 -0400
commit87e5216cd287f411b2ffab04dbc435f48c1d4aae (patch)
treed24346d6793233a2af6305f8463ae17765e00d44 /converter/config.py
parentf1f8e899c46de80d7f9fbb8f0b53983e18167bd2 (diff)
downloadtts-audiobook-generator-87e5216cd287f411b2ffab04dbc435f48c1d4aae.tar.gz
refactor: config.py simplified
Diffstat (limited to 'converter/config.py')
-rw-r--r--converter/config.py166
1 files changed, 23 insertions, 143 deletions
diff --git a/converter/config.py b/converter/config.py
index e5e5ff7..eb01462 100644
--- a/converter/config.py
+++ b/converter/config.py
@@ -1,161 +1,41 @@
"""Configuration for the audiobook converter.
-Edit these values to change the default voice and processing behavior.
-All paths are resolved relative to the project root, so the converter can
-be run from any working directory.
+Edit these values to change voices and processing behavior. Everything
+else (voice mode names, languages, speaker names, model ids, folders,
+file formats) is fixed in the code where it is used.
"""
-from pathlib import Path
+# Server endpoints. Voice clone needs the Base-model demo, which is a
+# separate server from the CustomVoice demo (that one only exposes
+# /run_instruct); the faster backend is the OpenAI-compatible server from
+# the faster-qwen3-tts repository.
+QWEN_API_URL = "http://127.0.0.1:7860" # CustomVoice demo
+VOICE_CLONE_API_URL = "http://127.0.0.1:7861" # Base-model demo
+FASTER_TTS_API_URL = "http://127.0.0.1:8000" # faster-qwen3-tts server
-# Project root (directory containing audiobook.py)
-BASE_DIR = Path(__file__).resolve().parent.parent
-
-# =============================================================================
-# VOICE MODES
-# =============================================================================
-
-VOICE_MODE_CUSTOM = "custom_voice"
-VOICE_MODE_CLONE = "voice_clone"
-VOICE_MODES = (VOICE_MODE_CUSTOM, VOICE_MODE_CLONE)
-
-# =============================================================================
-# TTS LANGUAGE SETTINGS
-# =============================================================================
-# Languages understood by the Qwen3-TTS API. The Gradio demo silently falls
-# back to "Auto" for unrecognized values, so languages are validated
-# client-side (see converter.tts.normalize_language) before reaching the API.
-# Display names must match the demo dropdown exactly.
-
-TTS_LANGUAGES = (
- "Auto",
- "Chinese",
- "English",
- "German",
- "Italian",
- "Portuguese",
- "Spanish",
- "Japanese",
- "Korean",
- "French",
- "Russian",
-)
-
-# Short aliases accepted on the command line (ISO 639-1 codes and common
-# shorthands), mapped to the display names above.
-TTS_LANGUAGE_ALIASES = {
- "zh": "Chinese",
- "en": "English",
- "de": "German",
- "it": "Italian",
- "pt": "Portuguese",
- "es": "Spanish",
- "ja": "Japanese",
- "ko": "Korean",
- "fr": "French",
- "ru": "Russian",
- "zh-cn": "Chinese",
- "zh-tw": "Chinese",
- "pt-br": "Portuguese",
- "en-us": "English",
- "en-gb": "English",
-}
-
-# =============================================================================
-# QWEN API CONFIGURATION
-# =============================================================================
+# Words per TTS generation request. Each API call is ONE model generation:
+# long generations lose prosody and can degrade into garbled audio.
+CHUNK_SIZE_WORDS = 40
+VOICE_CLONE_MAX_CHUNK_CHARS = 200 # Server-side re-chunking limit for clone requests
+VOICE_CLONE_CHUNK_GAP = 0 # Pause (seconds) between server-side clone chunks
+MIN_DELAY_BETWEEN_CHUNKS = 0 # Pause between API calls (rate-limit protection; local servers need none)
-QWEN_API_URL = "http://127.0.0.1:7860" # CustomVoice demo endpoint
API_TIMEOUT = 300 # Seconds before an API call times out
-MAX_RETRIES = 3 # Retry failed chunks
+MAX_RETRIES = 3 # Attempts per chunk request
+HEARTBEAT_INTERVAL_SECONDS = 30 # Print "still working" this often during a chunk
-# =============================================================================
-# CUSTOM VOICE SETTINGS (pre-built speakers, always uses the 1.7B model)
-# =============================================================================
+LANGUAGE = "English"
+SEED = -1
CUSTOM_VOICE_SPEAKER = "Vivian"
-CUSTOM_VOICE_LANGUAGE = "English"
CUSTOM_VOICE_INSTRUCT = "Speak naturally and clearly, as if reading a dramatic book to an adult audience."
-CUSTOM_VOICE_MODEL_SIZE = "1.7B"
-CUSTOM_VOICE_SEED = -1
-CUSTOM_VOICE_MODEL_ID = "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice"
-
-# Map canonical speaker names to the display names used by the qwen-tts Gradio demo.
-SPEAKER_DISPLAY_NAMES = {
- "ryan": "Ryan",
- "serena": "Serena",
- "vivian": "Vivian",
- "uncle_fu": "Uncle Fu",
- "aiden": "Aiden",
- "ono_anna": "Ono Anna",
- "sohee": "Sohee",
- "eric": "Eric",
- "dylan": "Dylan",
-}
-# =============================================================================
-# VOICE CLONE SETTINGS (clone a voice from a reference audio file)
-# =============================================================================
-# Voice clone requires the Base-model demo (Qwen3-TTS-12Hz-1.7B-Base), which
-# exposes /run_voice_clone. The CustomVoice demo only exposes /run_instruct, so
-# run the Base demo on a separate port and point this at it.
-
-VOICE_CLONE_LANGUAGE = "English"
VOICE_CLONE_USE_XVECTOR_ONLY = False
-VOICE_CLONE_MODEL_SIZE = "1.7B"
-VOICE_CLONE_MAX_CHUNK_CHARS = 200
-VOICE_CLONE_CHUNK_GAP = 0
-VOICE_CLONE_SEED = -1
-VOICE_CLONE_API_URL = "http://127.0.0.1:7861"
-
-# =============================================================================
-# FASTER TTS SETTINGS (optional --faster backend)
-# =============================================================================
-# --faster talks to the OpenAI-compatible server from faster-qwen3-tts
-# (examples/openai_server.py) instead of the qwen-tts Gradio demos. The
-# reference voice (ref audio, ref text) and language are configured on the
-# SERVER side (--ref-audio/--ref-text or a --voices JSON file); the converter
-# only sends text. See the "Faster backend" section of the README.
-FASTER_TTS_API_URL = "http://127.0.0.1:8000"
-# Voice entry to request. MUST match a key in the server's voices.json (or
-# "default" when the server was launched with --ref-audio). NOTE: the stock
-# server silently falls back to its first configured voice when the requested
-# name is unknown, so a mismatch here is easy to miss.
+# Must match a key in the faster server's voices.json ("default" when the
+# server was launched with --ref-audio). The server silently falls back to
+# its first voice for unknown names.
FASTER_TTS_VOICE = "default"
-FASTER_TTS_SAMPLE_RATE = 24000 # Qwen3-TTS 12Hz codec output rate
-# Safety net: the faster server takes one generation per request, so any
-# chunk longer than this is sub-chunked client-side (a no-op at the default
-# CHUNK_SIZE_WORDS above; kept in case the chunk size is ever raised).
-FASTER_SUBCHUNK_WORDS = 40 # ~200 chars per request
-FASTER_HTTP_TIMEOUT = 300 # Seconds before a speech request times out
-FASTER_SUBCHUNK_RETRIES = 3 # Attempts per sub-chunk request
-
-# =============================================================================
-# PROCESSING SETTINGS
-# =============================================================================
-
-BOOKS_FOLDER = BASE_DIR / "input" # Input folder
-AUDIOBOOKS_FOLDER = BASE_DIR / "output" # Output folder
-CHUNKS_FOLDER = BASE_DIR / "chunks" # Scratch space for per-chunk audio (cleaned per book)
-LOGS_FOLDER = BASE_DIR / "logs"
-DEBUG_FOLDER = BASE_DIR / "debug" # Per-chunk audio + text dumps for --debug (kept across runs)
-
-# Words per TTS generation request. Each API call is ONE model generation:
-# long generations lose prosody, can degrade into garbled audio, and text
-# past the model's token limit is never spoken. ~40 words (~200 chars) is
-# the per-request length the old qwen-tts demo enforced server-side.
-CHUNK_SIZE_WORDS = 40
-# Pause between API calls (rate-limit protection for hosted demos; a local
-# server needs no delay).
-MIN_DELAY_BETWEEN_CHUNKS = 0
-HEARTBEAT_INTERVAL_SECONDS = 30 # Print "still working" this often during a chunk
-
-# =============================================================================
-# AUDIO OUTPUT SETTINGS
-# =============================================================================
AUDIO_FORMAT = "m4b" # Default output container
-AUDIO_FORMATS = ("mp3", "m4b", "ogg", "flac")
AUDIO_BITRATE = "128k"
-
-SUPPORTED_FORMATS = [".txt", ".pdf", ".epub"]