aboutsummaryrefslogtreecommitdiff
path: root/app/backends/sglomni/configs/voxtral_tts.yaml
blob: b53f83adbd773e4416e2aabb5b9d21788f180dab (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
# Voxtral TTS 4B with a pinned AR-engine memory budget.
#
# The upstream pipeline (VoxtralTTSPipelineConfig) colocates its three stages
# (preprocessing -> tts_generation -> vocoder) in one process on GPU 0 and
# leaves the engine's sglang mem_fraction_static unset: the static pool
# (weights + KV cache) is auto-sized to nearly all free VRAM at boot. On a
# 24 GB card that leaves only tens of MiB free once the engine's CUDA graphs
# and the colocated vocoder are resident — the first /v1/audio/speech request
# aborts with "CUDA out of memory. Tried to allocate ~100 MiB" and every
# retry fails identically (the same failure the qwen3_tts/higgs/zonos2
# vendored configs fix).
#
# 0.70 pins the static pool at ~70% of the card (~16.5 GB on a 24 GB GPU —
# ~8 GB of weights leaves a KV pool far larger than any narration request
# needs) and leaves ~7 GB for the vocoder, CUDA graphs, transient
# allocations, and other GPU processes. Precedent: dots_tts.yaml pins this
# same knob. NOTE the stage name: Voxtral's engine stage is tts_generation,
# not tts_engine.
config_cls: VoxtralTTSPipelineConfig
model_path: mistralai/Voxtral-4B-TTS-2603

stages:
  tts_generation:
    engine:
      mem_fraction_static: 0.70