aboutsummaryrefslogtreecommitdiff
path: root/converter/config.py
diff options
context:
space:
mode:
Diffstat (limited to 'converter/config.py')
-rw-r--r--converter/config.py80
1 files changed, 80 insertions, 0 deletions
diff --git a/converter/config.py b/converter/config.py
new file mode 100644
index 0000000..1b203d2
--- /dev/null
+++ b/converter/config.py
@@ -0,0 +1,80 @@
+"""Configuration for the audiobook converter.
+
+Edit these values to change the default voice and processing behavior.
+All paths are resolved relative to the project root, so the converter can
+be run from any working directory.
+"""
+
+from pathlib import Path
+
+# Project root (directory containing audiobook_converter.py)
+BASE_DIR = Path(__file__).resolve().parent.parent
+
+# =============================================================================
+# QWEN API CONFIGURATION
+# =============================================================================
+
+QWEN_API_URL = "http://127.0.0.1:7860" # CustomVoice demo endpoint
+API_TIMEOUT = 300 # Seconds before an API call times out
+MAX_RETRIES = 3 # Retry failed chunks
+
+# =============================================================================
+# CUSTOM VOICE SETTINGS (pre-built speakers, always uses the 1.7B model)
+# =============================================================================
+
+CUSTOM_VOICE_SPEAKER = "Vivian"
+CUSTOM_VOICE_LANGUAGE = "English"
+CUSTOM_VOICE_INSTRUCT = "Speak naturally and clearly, as if reading a dramatic book to an adult audience."
+CUSTOM_VOICE_MODEL_SIZE = "1.7B"
+CUSTOM_VOICE_SEED = -1
+CUSTOM_VOICE_MODEL_ID = "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice"
+
+# Map canonical speaker names to the display names used by the qwen-tts Gradio demo.
+SPEAKER_DISPLAY_NAMES = {
+ "ryan": "Ryan",
+ "serena": "Serena",
+ "vivian": "Vivian",
+ "uncle_fu": "Uncle Fu",
+ "aiden": "Aiden",
+ "ono_anna": "Ono Anna",
+ "sohee": "Sohee",
+ "eric": "Eric",
+ "dylan": "Dylan",
+}
+
+# =============================================================================
+# VOICE CLONE SETTINGS (clone a voice from a reference audio file)
+# =============================================================================
+# Voice clone requires the Base-model demo (Qwen3-TTS-12Hz-1.7B-Base), which
+# exposes /run_voice_clone. The CustomVoice demo only exposes /run_instruct, so
+# run the Base demo on a separate port and point this at it.
+
+VOICE_CLONE_LANGUAGE = "English"
+VOICE_CLONE_USE_XVECTOR_ONLY = False
+VOICE_CLONE_MODEL_SIZE = "1.7B"
+VOICE_CLONE_MAX_CHUNK_CHARS = 200
+VOICE_CLONE_CHUNK_GAP = 0
+VOICE_CLONE_SEED = -1
+VOICE_CLONE_API_URL = "http://127.0.0.1:7861"
+
+# =============================================================================
+# PROCESSING SETTINGS
+# =============================================================================
+
+BOOKS_FOLDER = BASE_DIR / "book_to_convert" # Input folder
+AUDIOBOOKS_FOLDER = BASE_DIR / "audiobooks" # Output folder
+CHUNKS_FOLDER = BASE_DIR / "chunks" # Scratch space for per-chunk audio (cleaned per book)
+LOGS_FOLDER = BASE_DIR / "logs"
+
+CHUNK_SIZE_WORDS = 1500 # Words per TTS chunk
+MIN_DELAY_BETWEEN_CHUNKS = 1 # Seconds between API calls
+HEARTBEAT_INTERVAL_SECONDS = 30 # Print "still working" this often during a chunk
+
+# =============================================================================
+# AUDIO OUTPUT SETTINGS
+# =============================================================================
+
+AUDIO_FORMAT = "mp3"
+AUDIO_BITRATE = "128k"
+
+SUPPORTED_FORMATS = [".txt", ".pdf", ".epub"]