From 32b29d4c6c50591f966367e2ba1e9065a5c32e21 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Wed, 20 May 2026 17:43:42 -0500 Subject: [PATCH] fix: add deepgemm, update base image and hub for cuda 13.0 --- .runpod/hub.json | 12 +++++++++++- Dockerfile | 11 +++++++---- builder/requirements.txt | 2 +- docs/configuration.md | 3 +++ 4 files changed, 22 insertions(+), 6 deletions(-) diff --git a/.runpod/hub.json b/.runpod/hub.json index 6ab87fb..821dcf3 100644 --- a/.runpod/hub.json +++ b/.runpod/hub.json @@ -9,7 +9,7 @@ "containerDiskInGb": 150, "gpuIds": "ADA_80_PRO,AMPERE_80", "gpuCount": 1, - "allowedCudaVersions": ["12.9", "12.8"], + "allowedCudaVersions": ["13.0"], "presets": [ { "name": "deepseek-ai/deepseek-r1-distill-llama-8b", @@ -805,6 +805,16 @@ "default": "expandable_segments:True", "advanced": true } + }, + { + "key": "VLLM_USE_DEEP_GEMM", + "input": { + "name": "Use DeepGEMM", + "type": "boolean", + "description": "Enable DeepGEMM FP8 kernels (MoE and MQA logits). Disabled by default — enable on H100/H200 for potential throughput gains. Some GPUs (e.g. H20) may perform better with this off. Enabling also adds a warmup period on startup.", + "default": false, + "advanced": true + } } ] } diff --git a/Dockerfile b/Dockerfile index ddfcf27..79cbf19 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,7 +1,7 @@ -FROM nvidia/cuda:13.0.2-base-ubuntu22.04 +FROM nvidia/cuda:13.0.2-devel-ubuntu22.04 RUN apt-get update -y \ - && apt-get install -y python3-pip curl \ + && apt-get install -y python3-pip curl git \ && curl -LsSf https://astral.sh/uv/install.sh | sh ENV PATH="/root/.local/bin:$PATH" @@ -10,7 +10,8 @@ RUN ldconfig /usr/local/cuda-13.0/compat/ # Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.20.1" + uv pip install --system "vllm[flashinfer]==0.20.2" && \ + uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@v2.1.1.post3 --no-build-isolation # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) COPY builder/requirements.txt /requirements.txt @@ -42,7 +43,9 @@ ENV MODEL_NAME=$MODEL_NAME \ # Prevent rayon thread pool panic in containers where ulimit -u < nproc # (tokenizers uses Rust's rayon which tries to spawn threads = CPU cores) TOKENIZERS_PARALLELISM=false \ - RAYON_NUM_THREADS=4 + RAYON_NUM_THREADS=4 \ + # Disable DeepGEMM MoE kernels by default; override with VLLM_USE_DEEP_GEMM=1 to enable + VLLM_USE_DEEP_GEMM=0 ENV PYTHONPATH="/:/vllm-workspace" diff --git a/builder/requirements.txt b/builder/requirements.txt index f3ad976..19fceca 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -3,7 +3,7 @@ pandas pyarrow runpod==1.9.0 huggingface-hub -lmcache==0.4.2 +lmcache==0.4.5 packaging>=24.2 typing-extensions>=4.8.0 pydantic diff --git a/docs/configuration.md b/docs/configuration.md index fb21f9d..86fd586 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -97,10 +97,13 @@ If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When | `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. | | `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. | | `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. | +| `VLLM_USE_DEEP_GEMM` | `0` | `bool` | Enable DeepGEMM FP8 kernels for MoE and MQA logits computation. Disabled by default. See note below. | | `ATTENTION_BACKEND` | `None` | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `FLASHINFER`, `TRITON_FLASH_ATTN`). Replaces deprecated `VLLM_ATTENTION_BACKEND`. | | `ASYNC_SCHEDULING` | `None` | `bool` | Enable async scheduling (overlaps engine scheduling with GPU execution). Default: enabled in vLLM 0.14.0+. Set to `false` to disable. | | `STREAM_INTERVAL` | `1` | `int` | Controls how often to yield streaming results. Lower = more frequent updates. | +> **Note (`VLLM_USE_DEEP_GEMM`):** DeepGEMM is used in two places: MoE weight computation and MQA logits computation. It is necessary for MQA logits computation on supported hardware. Set `VLLM_USE_DEEP_GEMM=0` to disable the MoE part and fall back to flashinfer/cutlass FP8 kernels. Some users report better performance with `VLLM_USE_DEEP_GEMM=0`, particularly on H20 GPUs. Disabling it also skips the DeepGEMM warmup phase, reducing cold-start time. Requires CUDA 13.0+ and SM90+ (H100/H200) to use; the library is installed but inactive by default. + ## Tokenizer Settings | Variable | Default | Type/Choices | Description |