fix: update VLLM_USE_DEEP_GEMM hub to default to 0

This commit is contained in:
velaraptor-runpod
2026-05-27 11:26:52 -05:00
parent 6265b99348
commit 4c91f2c5b5
+3 -3
View File
@@ -810,9 +810,9 @@
"key": "VLLM_USE_DEEP_GEMM", "key": "VLLM_USE_DEEP_GEMM",
"input": { "input": {
"name": "Use DeepGEMM", "name": "Use DeepGEMM",
"type": "boolean", "type": "string",
"description": "Enable DeepGEMM FP8 kernels (MoE and MQA logits). Disabled by default — enable on H100/H200 for potential throughput gains. Some GPUs (e.g. H20) may perform better with this off. Enabling also adds a warmup period on startup.", "description": "Enable DeepGEMM FP8 kernels (MoE and MQA logits). Set to 1 to enable, 0 to disable. Required for DeepSeek V4 models. Disabled by default — enable on H100/H200 for potential throughput gains. Some GPUs (e.g. H20) may perform better with this off.",
"default": false, "default": "0",
"advanced": true "advanced": true
} }
} }