fix: add deepgemm, update base image and hub for cuda 13.0
This commit is contained in:
+11
-1
@@ -9,7 +9,7 @@
|
||||
"containerDiskInGb": 150,
|
||||
"gpuIds": "ADA_80_PRO,AMPERE_80",
|
||||
"gpuCount": 1,
|
||||
"allowedCudaVersions": ["12.9", "12.8"],
|
||||
"allowedCudaVersions": ["13.0"],
|
||||
"presets": [
|
||||
{
|
||||
"name": "deepseek-ai/deepseek-r1-distill-llama-8b",
|
||||
@@ -805,6 +805,16 @@
|
||||
"default": "expandable_segments:True",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "VLLM_USE_DEEP_GEMM",
|
||||
"input": {
|
||||
"name": "Use DeepGEMM",
|
||||
"type": "boolean",
|
||||
"description": "Enable DeepGEMM FP8 kernels (MoE and MQA logits). Disabled by default — enable on H100/H200 for potential throughput gains. Some GPUs (e.g. H20) may perform better with this off. Enabling also adds a warmup period on startup.",
|
||||
"default": false,
|
||||
"advanced": true
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
+7
-4
@@ -1,7 +1,7 @@
|
||||
FROM nvidia/cuda:13.0.2-base-ubuntu22.04
|
||||
FROM nvidia/cuda:13.0.2-devel-ubuntu22.04
|
||||
|
||||
RUN apt-get update -y \
|
||||
&& apt-get install -y python3-pip curl \
|
||||
&& apt-get install -y python3-pip curl git \
|
||||
&& curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||
|
||||
ENV PATH="/root/.local/bin:$PATH"
|
||||
@@ -10,7 +10,8 @@ RUN ldconfig /usr/local/cuda-13.0/compat/
|
||||
|
||||
# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
|
||||
RUN uv pip install --system "packaging>=24.2" && \
|
||||
uv pip install --system "vllm[flashinfer]==0.20.1"
|
||||
uv pip install --system "vllm[flashinfer]==0.20.2" && \
|
||||
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@v2.1.1.post3 --no-build-isolation
|
||||
|
||||
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
||||
COPY builder/requirements.txt /requirements.txt
|
||||
@@ -42,7 +43,9 @@ ENV MODEL_NAME=$MODEL_NAME \
|
||||
# Prevent rayon thread pool panic in containers where ulimit -u < nproc
|
||||
# (tokenizers uses Rust's rayon which tries to spawn threads = CPU cores)
|
||||
TOKENIZERS_PARALLELISM=false \
|
||||
RAYON_NUM_THREADS=4
|
||||
RAYON_NUM_THREADS=4 \
|
||||
# Disable DeepGEMM MoE kernels by default; override with VLLM_USE_DEEP_GEMM=1 to enable
|
||||
VLLM_USE_DEEP_GEMM=0
|
||||
|
||||
ENV PYTHONPATH="/:/vllm-workspace"
|
||||
|
||||
|
||||
@@ -3,7 +3,7 @@ pandas
|
||||
pyarrow
|
||||
runpod==1.9.0
|
||||
huggingface-hub
|
||||
lmcache==0.4.2
|
||||
lmcache==0.4.5
|
||||
packaging>=24.2
|
||||
typing-extensions>=4.8.0
|
||||
pydantic
|
||||
|
||||
@@ -97,10 +97,13 @@ If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When
|
||||
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
|
||||
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
|
||||
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. |
|
||||
| `VLLM_USE_DEEP_GEMM` | `0` | `bool` | Enable DeepGEMM FP8 kernels for MoE and MQA logits computation. Disabled by default. See note below. |
|
||||
| `ATTENTION_BACKEND` | `None` | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `FLASHINFER`, `TRITON_FLASH_ATTN`). Replaces deprecated `VLLM_ATTENTION_BACKEND`. |
|
||||
| `ASYNC_SCHEDULING` | `None` | `bool` | Enable async scheduling (overlaps engine scheduling with GPU execution). Default: enabled in vLLM 0.14.0+. Set to `false` to disable. |
|
||||
| `STREAM_INTERVAL` | `1` | `int` | Controls how often to yield streaming results. Lower = more frequent updates. |
|
||||
|
||||
> **Note (`VLLM_USE_DEEP_GEMM`):** DeepGEMM is used in two places: MoE weight computation and MQA logits computation. It is necessary for MQA logits computation on supported hardware. Set `VLLM_USE_DEEP_GEMM=0` to disable the MoE part and fall back to flashinfer/cutlass FP8 kernels. Some users report better performance with `VLLM_USE_DEEP_GEMM=0`, particularly on H20 GPUs. Disabling it also skips the DeepGEMM warmup phase, reducing cold-start time. Requires CUDA 13.0+ and SM90+ (H100/H200) to use; the library is installed but inactive by default.
|
||||
|
||||
## Tokenizer Settings
|
||||
|
||||
| Variable | Default | Type/Choices | Description |
|
||||
|
||||
Reference in New Issue
Block a user