aboutsummaryrefslogtreecommitdiff
path: root/converter/config.py
blob: d15efa56914346e1f8947e0c93c8a79af7e20f72 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
# Default output options
AUDIO_FORMAT = "m4b"
AUDIO_BITRATE = "128k"
LANGUAGE = "English"

API_TIMEOUT = 600  # Timeout per chunk request in seconds
MAX_RETRIES = 3    # Attempts per chunk request
HEARTBEAT_INTERVAL_SECONDS = 30  # Print "still working" in console logs every N seconds

# Words per TTS generation request (client-side chunking).
# The qwen and faster backends always chunk with this size
# The audio.cpp backend chunks long text itself, so this is ignored
# by default with that backend. Force chunking with --chunk
CHUNK_SIZE = 250

# Default TTS backend.
# audiocpp: audiocpp_server
# qwen: qwen-tts-demo
# faster: faster-qwen-tts
# The --backend CLI flag overrides this
BACKEND = "audiocpp"

###############################################################################
# BACKEND 1: qwen-tts-demo (qwen) options                                       #
###############################################################################

# There are different API URLs for CustomVoice and Base models so you can run both at once
QWEN_API_URL     = "http://127.0.0.1:7860"  # CustomVoice model
CLONE_API_URL    = "http://127.0.0.1:7861"  # Base model

# Custom voice options
SPEAKER = "Vivian"  #Vivian, Serena, Uncle_Fu, Dylan, Eric, Ryan, Aiden, Ono_Anna, Sohee
INSTRUCT = "Speak naturally and clearly, as if reading a dramatic book to an adult audience."

# Don't clone with transcription, only use x-vector-only cloning. Generally "worse"
XVECTOR_ONLY = False

# Randomization seed. -1 means randomize with every generation
# With SEED = -1 and CONSTANT_SEED = True, one random seed will be used for the entire audiobook.
# This may keep the voice slightly more consistent across chunk boundaries
SEED = -1
CONSTANT_SEED = False

###############################################################################
# BACKEND 2: faster-qwen-tts options                                          #
###############################################################################
FASTER_API_URL   = "http://127.0.0.1:8000"    # faster-qwen3-tts server (Base model only)

# Default voice if no --voice is passed
FASTER_VOICE = "default"

###############################################################################
# BACKEND 3: audio.cpp options                                                #
###############################################################################
AUDIOCPP_API_URL = "http://127.0.0.1:8080"  # audio.cpp audiocpp_server

# Model ids in the audio.cpp server.json config. AUDIOCPP_MODEL_ID may point
# at any TTS model entry the server hosts (qwen3_tts, higgs_audio_tts,
# voxcpm2, index_tts2, ...); the family is detected from the server at
# startup and adapts the request automatically. Only qwen3_tts has built-in
# speakers (speaker mode); every other family needs --voice with a
# server-side voice preset. For single-model servers, set
# AUDIOCPP_CLONE_MODEL_ID to the same id as AUDIOCPP_MODEL_ID (or leave it
# empty); for Qwen3-TTS it typically names a second entry with the Base
# (cloning) model. A multi-model server (one server.json hosting several
# lazily-loaded entries) does not need editing here: leave AUDIOCPP_MODEL_ID
# unset to auto-select when only one entry is hosted, or pick the entry per
# run with the --model CLI flag.
AUDIOCPP_MODEL_ID = "qwen"
AUDIOCPP_CLONE_MODEL_ID = "qwen-clone"