aboutsummaryrefslogtreecommitdiff
path: root/app/backends/sglomni/configs/qwen3_tts_1_7b.yaml
diff options
context:
space:
mode:
Diffstat (limited to 'app/backends/sglomni/configs/qwen3_tts_1_7b.yaml')
-rw-r--r--app/backends/sglomni/configs/qwen3_tts_1_7b.yaml21
1 files changed, 21 insertions, 0 deletions
diff --git a/app/backends/sglomni/configs/qwen3_tts_1_7b.yaml b/app/backends/sglomni/configs/qwen3_tts_1_7b.yaml
index 4f7706d..b9cb5ec 100644
--- a/app/backends/sglomni/configs/qwen3_tts_1_7b.yaml
+++ b/app/backends/sglomni/configs/qwen3_tts_1_7b.yaml
@@ -1,2 +1,23 @@
+# Qwen3-TTS 1.7B Base with a pinned AR-engine memory budget.
+#
+# The upstream pipeline (Qwen3TTSPipelineConfig) colocates its three stages
+# (preprocessing -> tts_engine -> vocoder) in one process on GPU 0 and leaves
+# the engine's sglang mem_fraction_static unset: the static pool (weights +
+# KV cache) is auto-sized to nearly all free VRAM at boot. On a 24 GB card
+# that leaves only tens of MiB free once the engine's CUDA graphs (~1.5 GB)
+# and the colocated vocoder are resident — the first /v1/audio/speech request
+# aborts with "CUDA out of memory. Tried to allocate ~100 MiB" and every
+# retry fails identically (the weights themselves are small: 1.7B bf16).
+#
+# 0.70 pins the static pool at ~70% of the card (~16.5 GB on a 24 GB GPU —
+# a KV pool far larger than any narration request needs) and leaves ~7 GB
+# for the vocoder, CUDA graphs, transient allocations, and other GPU
+# processes. Precedent: dots_tts.yaml pins this same knob; zonos2_bf16.yaml
+# and higgs_audio_v3_tts.yaml fix the same 24 GB OOM class their way.
config_cls: Qwen3TTSPipelineConfig
model_path: Qwen/Qwen3-TTS-12Hz-1.7B-Base
+
+stages:
+ tts_engine:
+ engine:
+ mem_fraction_static: 0.70