From 22356ee2b3407074c28bbcae8a543c2eafca43db Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 30 Apr 2026 18:39:07 -0500 Subject: [PATCH 01/10] feat: upgrade vLLM to 0.20.0 - Bump vllm[flashinfer] to 0.20.0 in Dockerfile - Remove io_processor param from OpenAIServingRender (dropped in 0.20.0) Co-Authored-By: Claude Sonnet 4.6 --- Dockerfile | 2 +- src/engine.py | 1 - 2 files changed, 1 insertion(+), 2 deletions(-) diff --git a/Dockerfile b/Dockerfile index 1f05b31..6e55e64 100644 --- a/Dockerfile +++ b/Dockerfile @@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/ # Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.19.1" --extra-index-url https://download.pytorch.org/whl/cu129 + uv pip install --system "vllm[flashinfer]==0.20.0" --extra-index-url https://download.pytorch.org/whl/cu129 # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) COPY builder/requirements.txt /requirements.txt diff --git a/src/engine.py b/src/engine.py index 80bf72b..0fa60c9 100644 --- a/src/engine.py +++ b/src/engine.py @@ -285,7 +285,6 @@ class OpenAIvLLMEngine(vLLMEngine): self.openai_serving_render = OpenAIServingRender( model_config=self.llm.model_config, renderer=self.llm.renderer, - io_processor=self.llm.io_processor, model_registry=self.serving_models.registry, request_logger=None, chat_template=chat_template, From 747cdf589111fcdd12faecf216c055f94fd8d4df Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 30 Apr 2026 20:26:23 -0500 Subject: [PATCH 02/10] chore: update readme vllm version --- .runpod/README.md | 3 +++ README.md | 2 +- 2 files changed, 4 insertions(+), 1 deletion(-) diff --git a/.runpod/README.md b/.runpod/README.md index 6eb8652..b813ab8 100644 --- a/.runpod/README.md +++ b/.runpod/README.md @@ -6,6 +6,9 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API [![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) +Current vLLM version: [0.20.0](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) + + --- ## Endpoint Configuration diff --git a/README.md b/README.md index 0c7f0e5..7426fcf 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https: ![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png) -Current vLLM version: [0.16.0](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) +Current vLLM version: [0.20.0](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) > Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep) From 678bb4be8f3f3299b3784d46ac91b3fe12c3662e Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 7 May 2026 11:58:22 -0500 Subject: [PATCH 03/10] feat: update to 0.20.1 for patch fixes --- Dockerfile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Dockerfile b/Dockerfile index 6e55e64..66227cb 100644 --- a/Dockerfile +++ b/Dockerfile @@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/ # Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.20.0" --extra-index-url https://download.pytorch.org/whl/cu129 + uv pip install --system "vllm[flashinfer]==0.20.1" --extra-index-url https://download.pytorch.org/whl/cu129 # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) COPY builder/requirements.txt /requirements.txt From 9c139e8ceb346a3e57deb833048ad440a025d020 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Fri, 15 May 2026 11:35:53 -0400 Subject: [PATCH 04/10] update: update to 0.20.1, update dockerfile to cuda 13 --- .runpod/README.md | 2 +- Dockerfile | 4 ++-- README.md | 2 +- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/.runpod/README.md b/.runpod/README.md index 18e4369..390d4c2 100644 --- a/.runpod/README.md +++ b/.runpod/README.md @@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API [![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) -Current vLLM version: [0.20.0](https://github.com/vllm-project/vllm/releases/tag/v0.20.0) +Current vLLM version: [0.20.1](https://github.com/vllm-project/vllm/releases/tag/v0.20.1) --- diff --git a/Dockerfile b/Dockerfile index 66227cb..7005291 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,4 +1,4 @@ -FROM nvidia/cuda:12.9.1-base-ubuntu22.04 +FROM nvidia/cuda:13.0.2-base-ubuntu22.04 RUN apt-get update -y \ && apt-get install -y python3-pip curl \ @@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/ # Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.20.1" --extra-index-url https://download.pytorch.org/whl/cu129 + uv pip install --system "vllm[flashinfer]==0.20.1" --extra-index-url https://download.pytorch.org/whl/cu130 # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) COPY builder/requirements.txt /requirements.txt diff --git a/README.md b/README.md index 42dd66d..e25a02f 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https: ![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png) -Current vLLM version: [0.20.0](https://github.com/vllm-project/vllm/releases/tag/v0.20.0) +Current vLLM version: [0.20.1](https://github.com/vllm-project/vllm/releases/tag/v0.20.1) > Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep) From dcea4fc4f90ca9efa47df8c544c19f60d4f95f7c Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Fri, 15 May 2026 12:08:31 -0400 Subject: [PATCH 05/10] fix dockerfile --- Dockerfile | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/Dockerfile b/Dockerfile index 7005291..ddfcf27 100644 --- a/Dockerfile +++ b/Dockerfile @@ -6,11 +6,11 @@ RUN apt-get update -y \ ENV PATH="/root/.local/bin:$PATH" -RUN ldconfig /usr/local/cuda-12.9/compat/ +RUN ldconfig /usr/local/cuda-13.0/compat/ -# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels +# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.20.1" --extra-index-url https://download.pytorch.org/whl/cu130 + uv pip install --system "vllm[flashinfer]==0.20.1" # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) COPY builder/requirements.txt /requirements.txt From 32b29d4c6c50591f966367e2ba1e9065a5c32e21 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Wed, 20 May 2026 17:43:42 -0500 Subject: [PATCH 06/10] fix: add deepgemm, update base image and hub for cuda 13.0 --- .runpod/hub.json | 12 +++++++++++- Dockerfile | 11 +++++++---- builder/requirements.txt | 2 +- docs/configuration.md | 3 +++ 4 files changed, 22 insertions(+), 6 deletions(-) diff --git a/.runpod/hub.json b/.runpod/hub.json index 6ab87fb..821dcf3 100644 --- a/.runpod/hub.json +++ b/.runpod/hub.json @@ -9,7 +9,7 @@ "containerDiskInGb": 150, "gpuIds": "ADA_80_PRO,AMPERE_80", "gpuCount": 1, - "allowedCudaVersions": ["12.9", "12.8"], + "allowedCudaVersions": ["13.0"], "presets": [ { "name": "deepseek-ai/deepseek-r1-distill-llama-8b", @@ -805,6 +805,16 @@ "default": "expandable_segments:True", "advanced": true } + }, + { + "key": "VLLM_USE_DEEP_GEMM", + "input": { + "name": "Use DeepGEMM", + "type": "boolean", + "description": "Enable DeepGEMM FP8 kernels (MoE and MQA logits). Disabled by default — enable on H100/H200 for potential throughput gains. Some GPUs (e.g. H20) may perform better with this off. Enabling also adds a warmup period on startup.", + "default": false, + "advanced": true + } } ] } diff --git a/Dockerfile b/Dockerfile index ddfcf27..79cbf19 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,7 +1,7 @@ -FROM nvidia/cuda:13.0.2-base-ubuntu22.04 +FROM nvidia/cuda:13.0.2-devel-ubuntu22.04 RUN apt-get update -y \ - && apt-get install -y python3-pip curl \ + && apt-get install -y python3-pip curl git \ && curl -LsSf https://astral.sh/uv/install.sh | sh ENV PATH="/root/.local/bin:$PATH" @@ -10,7 +10,8 @@ RUN ldconfig /usr/local/cuda-13.0/compat/ # Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.20.1" + uv pip install --system "vllm[flashinfer]==0.20.2" && \ + uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@v2.1.1.post3 --no-build-isolation # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) COPY builder/requirements.txt /requirements.txt @@ -42,7 +43,9 @@ ENV MODEL_NAME=$MODEL_NAME \ # Prevent rayon thread pool panic in containers where ulimit -u < nproc # (tokenizers uses Rust's rayon which tries to spawn threads = CPU cores) TOKENIZERS_PARALLELISM=false \ - RAYON_NUM_THREADS=4 + RAYON_NUM_THREADS=4 \ + # Disable DeepGEMM MoE kernels by default; override with VLLM_USE_DEEP_GEMM=1 to enable + VLLM_USE_DEEP_GEMM=0 ENV PYTHONPATH="/:/vllm-workspace" diff --git a/builder/requirements.txt b/builder/requirements.txt index f3ad976..19fceca 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -3,7 +3,7 @@ pandas pyarrow runpod==1.9.0 huggingface-hub -lmcache==0.4.2 +lmcache==0.4.5 packaging>=24.2 typing-extensions>=4.8.0 pydantic diff --git a/docs/configuration.md b/docs/configuration.md index fb21f9d..86fd586 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -97,10 +97,13 @@ If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When | `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. | | `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. | | `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. | +| `VLLM_USE_DEEP_GEMM` | `0` | `bool` | Enable DeepGEMM FP8 kernels for MoE and MQA logits computation. Disabled by default. See note below. | | `ATTENTION_BACKEND` | `None` | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `FLASHINFER`, `TRITON_FLASH_ATTN`). Replaces deprecated `VLLM_ATTENTION_BACKEND`. | | `ASYNC_SCHEDULING` | `None` | `bool` | Enable async scheduling (overlaps engine scheduling with GPU execution). Default: enabled in vLLM 0.14.0+. Set to `false` to disable. | | `STREAM_INTERVAL` | `1` | `int` | Controls how often to yield streaming results. Lower = more frequent updates. | +> **Note (`VLLM_USE_DEEP_GEMM`):** DeepGEMM is used in two places: MoE weight computation and MQA logits computation. It is necessary for MQA logits computation on supported hardware. Set `VLLM_USE_DEEP_GEMM=0` to disable the MoE part and fall back to flashinfer/cutlass FP8 kernels. Some users report better performance with `VLLM_USE_DEEP_GEMM=0`, particularly on H20 GPUs. Disabling it also skips the DeepGEMM warmup phase, reducing cold-start time. Requires CUDA 13.0+ and SM90+ (H100/H200) to use; the library is installed but inactive by default. + ## Tokenizer Settings | Variable | Default | Type/Choices | Description | From 69968a6b395c9170800fdbeaa4ae783bbadfc0ad Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 21 May 2026 15:34:09 -0500 Subject: [PATCH 07/10] chore: fix logging, warning for text prompt --- src/engine.py | 50 ++++++++++++++++++++++++++++++++---------------- src/tokenizer.py | 6 ++++-- 2 files changed, 38 insertions(+), 18 deletions(-) diff --git a/src/engine.py b/src/engine.py index 0fa60c9..6fd321d 100644 --- a/src/engine.py +++ b/src/engine.py @@ -1,4 +1,5 @@ import asyncio +import inspect import json import logging import os @@ -7,6 +8,7 @@ from typing import AsyncGenerator, Optional from dotenv import load_dotenv from vllm import AsyncLLMEngine +from vllm.inputs import TextPrompt from vllm.entrypoints.logger import RequestLogger from vllm.entrypoints.anthropic.protocol import AnthropicMessagesRequest, AnthropicMessagesResponse, AnthropicError, AnthropicErrorResponse from vllm.entrypoints.anthropic.serving import AnthropicServingMessages @@ -30,20 +32,33 @@ class vLLMEngine: def __init__(self, engine = None): load_dotenv() # For local development self.engine_args = get_engine_args() - logging.info(f"Engine args: {self.engine_args}") - - # Initialize vLLM engine first - self.llm = self._initialize_llm() if engine is None else engine.llm - - # Only create custom tokenizer wrapper if not using mistral tokenizer mode - # For mistral models, let vLLM handle tokenizer initialization - if self.engine_args.tokenizer_mode != 'mistral': - self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model, - self.engine_args.tokenizer_revision, - self.engine_args.trust_remote_code) + + if engine is None: + ea = self.engine_args + summary = { + "model": ea.model, + "dtype": ea.dtype, + "quantization": ea.quantization, + "max_model_len": ea.max_model_len, + "tensor_parallel_size": ea.tensor_parallel_size, + "gpu_memory_utilization": ea.gpu_memory_utilization, + } + if ea.tokenizer and ea.tokenizer != ea.model: + summary["tokenizer"] = ea.tokenizer + logging.info("Engine config: %s", summary) + logging.debug("Full engine args: %s", ea) + + self.llm = self._initialize_llm() + + if self.engine_args.tokenizer_mode != 'mistral': + self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model, + self.engine_args.tokenizer_revision, + self.engine_args.trust_remote_code) + else: + self.tokenizer = None else: - # For mistral models, we'll get the tokenizer from vLLM later - self.tokenizer = None + self.llm = engine.llm + self.tokenizer = engine.tokenizer self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY)) self.default_batch_size = int(os.getenv("DEFAULT_BATCH_SIZE", DEFAULT_BATCH_SIZE)) @@ -116,7 +131,7 @@ class vLLMEngine: if apply_chat_template or isinstance(llm_input, list): tokenizer_wrapper = self._get_tokenizer_for_chat_template() llm_input = tokenizer_wrapper.apply_chat_template(llm_input) - results_generator = self.llm.generate(llm_input, validated_sampling_params, request_id) + results_generator = self.llm.generate(TextPrompt(prompt=llm_input), validated_sampling_params, request_id) n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0} @@ -356,8 +371,11 @@ class OpenAIvLLMEngine(vLLMEngine): enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true', ) - if hasattr(self.chat_engine, 'warmup'): - await self.chat_engine.warmup() + warmup = getattr(self.chat_engine, 'warmup', None) + if callable(warmup): + result = warmup() + if inspect.isawaitable(result): + await result async def generate(self, openai_request: JobInput): # Ensure engines are ready (no-op if already initialized at startup) diff --git a/src/tokenizer.py b/src/tokenizer.py index b7b866b..7dbd007 100644 --- a/src/tokenizer.py +++ b/src/tokenizer.py @@ -1,10 +1,12 @@ -from transformers import AutoTokenizer +import logging import os from typing import Union +from transformers import AutoTokenizer + class TokenizerWrapper: def __init__(self, tokenizer_name_or_path, tokenizer_revision, trust_remote_code): - print(f"tokenizer_name_or_path: {tokenizer_name_or_path}, tokenizer_revision: {tokenizer_revision}, trust_remote_code: {trust_remote_code}") + logging.debug("tokenizer_name_or_path: %s, tokenizer_revision: %s, trust_remote_code: %s", tokenizer_name_or_path, tokenizer_revision, trust_remote_code) self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_name_or_path, revision=tokenizer_revision or "main", trust_remote_code=trust_remote_code) self.custom_chat_template = os.getenv("CUSTOM_CHAT_TEMPLATE") self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template) From c2e6cc9f61aad3be5ff7daf50b72cb9467b199bc Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 21 May 2026 15:39:19 -0500 Subject: [PATCH 08/10] chore: update readme with correct vllm version --- .runpod/README.md | 2 +- README.md | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/.runpod/README.md b/.runpod/README.md index 390d4c2..d9e63d4 100644 --- a/.runpod/README.md +++ b/.runpod/README.md @@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API [![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) -Current vLLM version: [0.20.1](https://github.com/vllm-project/vllm/releases/tag/v0.20.1) +Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2) --- diff --git a/README.md b/README.md index e25a02f..2e5a1e8 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https: ![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png) -Current vLLM version: [0.20.1](https://github.com/vllm-project/vllm/releases/tag/v0.20.1) +Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2) > Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep) From da01193a3d40a65a18c9c1d6d6a844fd3cad331a Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 21 May 2026 15:57:40 -0500 Subject: [PATCH 09/10] chore: fix readme --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index 2e5a1e8..1af11b3 100644 --- a/README.md +++ b/README.md @@ -47,7 +47,7 @@ Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag **📦 Docker Image**: `runpod/worker-v1-vllm:` - **Available Versions**: See [GitHub Releases](https://github.com/runpod-workers/worker-vllm/releases) -- **CUDA Compatibility**: Requires CUDA >= 12.1 +- **CUDA Compatibility**: Requires CUDA >= 13.0 ### Configuration From 026f8d700bfd4c6ec706757f54a827c430cf66bd Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 21 May 2026 18:46:14 -0500 Subject: [PATCH 10/10] fix: specify deepgemm commit version --- Dockerfile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Dockerfile b/Dockerfile index 79cbf19..5e4efd1 100644 --- a/Dockerfile +++ b/Dockerfile @@ -11,7 +11,7 @@ RUN ldconfig /usr/local/cuda-13.0/compat/ # Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ uv pip install --system "vllm[flashinfer]==0.20.2" && \ - uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@v2.1.1.post3 --no-build-isolation + uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) COPY builder/requirements.txt /requirements.txt