From 4c91f2c5b58e449480a96d8d1cbefcb7e729156b Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Wed, 27 May 2026 11:26:52 -0500 Subject: [PATCH 1/2] fix: update VLLM_USE_DEEP_GEMM hub to default to 0 --- .runpod/hub.json | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/.runpod/hub.json b/.runpod/hub.json index 821dcf3..4dd7b8e 100644 --- a/.runpod/hub.json +++ b/.runpod/hub.json @@ -810,9 +810,9 @@ "key": "VLLM_USE_DEEP_GEMM", "input": { "name": "Use DeepGEMM", - "type": "boolean", - "description": "Enable DeepGEMM FP8 kernels (MoE and MQA logits). Disabled by default — enable on H100/H200 for potential throughput gains. Some GPUs (e.g. H20) may perform better with this off. Enabling also adds a warmup period on startup.", - "default": false, + "type": "string", + "description": "Enable DeepGEMM FP8 kernels (MoE and MQA logits). Set to 1 to enable, 0 to disable. Required for DeepSeek V4 models. Disabled by default — enable on H100/H200 for potential throughput gains. Some GPUs (e.g. H20) may perform better with this off.", + "default": "0", "advanced": true } } From 9edc5715ce7e5beefea48e450ec7620fe1de6c78 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Wed, 27 May 2026 11:33:37 -0500 Subject: [PATCH 2/2] fix: update configuration.md --- docs/configuration.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/configuration.md b/docs/configuration.md index 86fd586..b15b927 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -97,12 +97,12 @@ If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When | `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. | | `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. | | `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. | -| `VLLM_USE_DEEP_GEMM` | `0` | `bool` | Enable DeepGEMM FP8 kernels for MoE and MQA logits computation. Disabled by default. See note below. | +| `VLLM_USE_DEEP_GEMM` | `0` | `str` (`0`/`1`) | Enable DeepGEMM FP8 kernels for MoE and MQA logits computation. Disabled by default. Must be `"0"` or `"1"` — not `true`/`false`. See note below. | | `ATTENTION_BACKEND` | `None` | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `FLASHINFER`, `TRITON_FLASH_ATTN`). Replaces deprecated `VLLM_ATTENTION_BACKEND`. | | `ASYNC_SCHEDULING` | `None` | `bool` | Enable async scheduling (overlaps engine scheduling with GPU execution). Default: enabled in vLLM 0.14.0+. Set to `false` to disable. | | `STREAM_INTERVAL` | `1` | `int` | Controls how often to yield streaming results. Lower = more frequent updates. | -> **Note (`VLLM_USE_DEEP_GEMM`):** DeepGEMM is used in two places: MoE weight computation and MQA logits computation. It is necessary for MQA logits computation on supported hardware. Set `VLLM_USE_DEEP_GEMM=0` to disable the MoE part and fall back to flashinfer/cutlass FP8 kernels. Some users report better performance with `VLLM_USE_DEEP_GEMM=0`, particularly on H20 GPUs. Disabling it also skips the DeepGEMM warmup phase, reducing cold-start time. Requires CUDA 13.0+ and SM90+ (H100/H200) to use; the library is installed but inactive by default. +> **Note (`VLLM_USE_DEEP_GEMM`):** DeepGEMM is used in two places: MoE weight computation and MQA logits computation. It is necessary for MQA logits computation on supported hardware — required for DeepSeek V4 models. Set `VLLM_USE_DEEP_GEMM=1` to enable. Set `VLLM_USE_DEEP_GEMM=0` to disable the MoE part and fall back to flashinfer/cutlass FP8 kernels. **Value must be `"0"` or `"1"` — not `"true"`/`"false"`.** Some users report better performance with `VLLM_USE_DEEP_GEMM=0`, particularly on H20 GPUs. Disabling it also skips the DeepGEMM warmup phase, reducing cold-start time. Requires CUDA 13.0+ and SM90+ (H100/H200) to use; the library is installed but inactive by default. ## Tokenizer Settings