fix: update VLLM_USE_DEEP_GEMM hub to default to 0
This commit is contained in:
+3
-3
@@ -810,9 +810,9 @@
|
|||||||
"key": "VLLM_USE_DEEP_GEMM",
|
"key": "VLLM_USE_DEEP_GEMM",
|
||||||
"input": {
|
"input": {
|
||||||
"name": "Use DeepGEMM",
|
"name": "Use DeepGEMM",
|
||||||
"type": "boolean",
|
"type": "string",
|
||||||
"description": "Enable DeepGEMM FP8 kernels (MoE and MQA logits). Disabled by default — enable on H100/H200 for potential throughput gains. Some GPUs (e.g. H20) may perform better with this off. Enabling also adds a warmup period on startup.",
|
"description": "Enable DeepGEMM FP8 kernels (MoE and MQA logits). Set to 1 to enable, 0 to disable. Required for DeepSeek V4 models. Disabled by default — enable on H100/H200 for potential throughput gains. Some GPUs (e.g. H20) may perform better with this off.",
|
||||||
"default": false,
|
"default": "0",
|
||||||
"advanced": true
|
"advanced": true
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user