Compare commits

...
20 Commits
Author SHA1 Message Date
Tim PietruskyandGitHub dac05b62b3 fix: make .runpod/tests.json hub tests pass on CUDA 13.0 (#299)
Release / release (push) Waiting to run
Three coupled fixes verified end-to-end on a private fork
(TimPietruskyRunPod/worker-vllm v0.1.3 → both hub tests passing):

1. tests.json allowedCudaVersions: 12.x → 13.0
   The Dockerfile and hub.json moved to CUDA 13.0 in v2.20.0
   (#288, #289), but tests.json was still pinned to 12.5–12.9, so
   the test pod was scheduled on a GPU with driver < 13.0 and
   container init failed at the nvidia-container-cli hook with
   "unsatisfied condition: cuda>=13.0".

2. requirements.txt kernels<0.15
   huggingface/kernels v0.15.1 tightened LayerRepository to require
   a revision or version argument
   (https://github.com/huggingface/kernels/pull/544). transformers
   >=5 still constructs LayerRepository(repo_id=..., layer_name=...)
   without either, so worker import raised ValueError during
   `from transformers import ...`. 0.14.1 is the last safe release.

3. tests.json timeout 30000 → 300000
   vLLM cold start (torch.compile + FlashInfer warmup) on RTX 4090
   for SmolLM2-135M takes ~60–70s before the first request can be
   served. The previous 30s per-test timeout fired before the
   worker came up, producing "context cancelled or timed out:
   context deadline exceeded" for every test even when the worker
   was healthy. 300s gives enough headroom for cold start + the
   actual inference call.

Refs: DR-1161
2026-06-02 17:08:54 +02:00
Tim PietruskyandGitHub 14b74a4989 chore: re-enable .runpod/tests.json hub tests (#295)
Release / release (push) Waiting to run
Rename tests_json back to tests.json to re-enable the automated hub
tests that were temporarily disabled in #253.

Refs: DR-1161
2026-06-01 17:52:23 +02:00
chrisvelaandGitHub 50aba8fb57 Merge pull request #293 from runpod-workers/fix/update-deep-gemm-hub-value
Release / release (push) Waiting to run
fix: update VLLM_USE_DEEP_GEMM hub to default to 0
2026-05-27 11:51:48 -05:00
velaraptor-runpod 9edc5715ce fix: update configuration.md 2026-05-27 11:33:37 -05:00
velaraptor-runpod 4c91f2c5b5 fix: update VLLM_USE_DEEP_GEMM hub to default to 0 2026-05-27 11:26:52 -05:00
chrisvelaandGitHub 6265b99348 Merge pull request #288 from runpod-workers/feat/0.20.0
feat: update to 0.20.2
2026-05-26 17:17:45 -05:00
velaraptor-runpod 026f8d700b fix: specify deepgemm commit version 2026-05-21 18:46:54 -05:00
chrisvelaandGitHub 146bdb0252 Merge branch 'main' into feat/0.20.0 2026-05-21 16:04:39 -05:00
velaraptor-runpod da01193a3d chore: fix readme 2026-05-21 15:57:40 -05:00
velaraptor-runpod c2e6cc9f61 chore: update readme with correct vllm version 2026-05-21 15:39:19 -05:00
velaraptor-runpod 69968a6b39 chore: fix logging, warning for text prompt 2026-05-21 15:34:09 -05:00
velaraptor-runpod 32b29d4c6c fix: add deepgemm, update base image and hub for cuda 13.0 2026-05-20 17:43:42 -05:00
velaraptor-runpod dcea4fc4f9 fix dockerfile 2026-05-15 12:08:31 -04:00
velaraptor-runpod 9c139e8ceb update: update to 0.20.1, update dockerfile to cuda 13 2026-05-15 11:35:53 -04:00
velaraptor-runpod 678bb4be8f feat: update to 0.20.1 for patch fixes 2026-05-07 11:58:22 -05:00
chrisvelaandGitHub 87d7365126 Merge pull request #292 from runpod-workers/fix/open-ai
Release / release (push) Waiting to run
fix: fix warmup
2026-05-01 18:06:11 -05:00
velaraptor-runpod 0e83616f93 fix: fix warmup 2026-05-01 17:39:51 -05:00
velaraptor-runpod ed315a175e merge main 2026-05-01 15:28:05 -05:00
velaraptor-runpod 747cdf5891 chore: update readme vllm version 2026-04-30 20:26:23 -05:00
velaraptor-runpodandClaude Sonnet 4.6 22356ee2b3 feat: upgrade vLLM to 0.20.0
- Bump vllm[flashinfer] to 0.20.0 in Dockerfile
- Remove io_processor param from OpenAIServingRender (dropped in 0.20.0)

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-04-30 18:39:07 -05:00
9 changed files with 69 additions and 34 deletions
+1 -1
View File
@@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
[![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) [![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.19.1) Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
--- ---
+11 -1
View File
@@ -9,7 +9,7 @@
"containerDiskInGb": 150, "containerDiskInGb": 150,
"gpuIds": "ADA_80_PRO,AMPERE_80", "gpuIds": "ADA_80_PRO,AMPERE_80",
"gpuCount": 1, "gpuCount": 1,
"allowedCudaVersions": ["12.9", "12.8"], "allowedCudaVersions": ["13.0"],
"presets": [ "presets": [
{ {
"name": "deepseek-ai/deepseek-r1-distill-llama-8b", "name": "deepseek-ai/deepseek-r1-distill-llama-8b",
@@ -805,6 +805,16 @@
"default": "expandable_segments:True", "default": "expandable_segments:True",
"advanced": true "advanced": true
} }
},
{
"key": "VLLM_USE_DEEP_GEMM",
"input": {
"name": "Use DeepGEMM",
"type": "string",
"description": "Enable DeepGEMM FP8 kernels (MoE and MQA logits). Set to 1 to enable, 0 to disable. Required for DeepSeek V4 models. Disabled by default — enable on H100/H200 for potential throughput gains. Some GPUs (e.g. H20) may perform better with this off.",
"default": "0",
"advanced": true
}
} }
] ]
} }
+3 -3
View File
@@ -5,7 +5,7 @@
"input": { "input": {
"prompt": "Write a short poem about artificial intelligence." "prompt": "Write a short poem about artificial intelligence."
}, },
"timeout": 30000 "timeout": 300000
}, },
{ {
"name": "openai_messages_test", "name": "openai_messages_test",
@@ -26,7 +26,7 @@
"temperature": 0.1 "temperature": 0.1
} }
}, },
"timeout": 30000 "timeout": 300000
} }
], ],
"config": { "config": {
@@ -38,6 +38,6 @@
"value": "HuggingFaceTB/SmolLM2-135M-Instruct" "value": "HuggingFaceTB/SmolLM2-135M-Instruct"
} }
], ],
"allowedCudaVersions": ["12.9", "12.8", "12.7", "12.6", "12.5"] "allowedCudaVersions": ["13.0"]
} }
} }
+9 -6
View File
@@ -1,16 +1,17 @@
FROM nvidia/cuda:12.9.1-base-ubuntu22.04 FROM nvidia/cuda:13.0.2-devel-ubuntu22.04
RUN apt-get update -y \ RUN apt-get update -y \
&& apt-get install -y python3-pip curl \ && apt-get install -y python3-pip curl git \
&& curl -LsSf https://astral.sh/uv/install.sh | sh && curl -LsSf https://astral.sh/uv/install.sh | sh
ENV PATH="/root/.local/bin:$PATH" ENV PATH="/root/.local/bin:$PATH"
RUN ldconfig /usr/local/cuda-12.9/compat/ RUN ldconfig /usr/local/cuda-13.0/compat/
# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels # Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
RUN uv pip install --system "packaging>=24.2" && \ RUN uv pip install --system "packaging>=24.2" && \
uv pip install --system "vllm[flashinfer]==0.19.1" --extra-index-url https://download.pytorch.org/whl/cu129 uv pip install --system "vllm[flashinfer]==0.20.2" && \
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
COPY builder/requirements.txt /requirements.txt COPY builder/requirements.txt /requirements.txt
@@ -42,7 +43,9 @@ ENV MODEL_NAME=$MODEL_NAME \
# Prevent rayon thread pool panic in containers where ulimit -u < nproc # Prevent rayon thread pool panic in containers where ulimit -u < nproc
# (tokenizers uses Rust's rayon which tries to spawn threads = CPU cores) # (tokenizers uses Rust's rayon which tries to spawn threads = CPU cores)
TOKENIZERS_PARALLELISM=false \ TOKENIZERS_PARALLELISM=false \
RAYON_NUM_THREADS=4 RAYON_NUM_THREADS=4 \
# Disable DeepGEMM MoE kernels by default; override with VLLM_USE_DEEP_GEMM=1 to enable
VLLM_USE_DEEP_GEMM=0
ENV PYTHONPATH="/:/vllm-workspace" ENV PYTHONPATH="/:/vllm-workspace"
+2 -2
View File
@@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png) ![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png)
Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.19.1) Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep) > Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
@@ -47,7 +47,7 @@ Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag
**📦 Docker Image**: `runpod/worker-v1-vllm:<version>` **📦 Docker Image**: `runpod/worker-v1-vllm:<version>`
- **Available Versions**: See [GitHub Releases](https://github.com/runpod-workers/worker-vllm/releases) - **Available Versions**: See [GitHub Releases](https://github.com/runpod-workers/worker-vllm/releases)
- **CUDA Compatibility**: Requires CUDA >= 12.1 - **CUDA Compatibility**: Requires CUDA >= 13.0
### Configuration ### Configuration
+2 -2
View File
@@ -3,7 +3,7 @@ pandas
pyarrow pyarrow
runpod==1.9.0 runpod==1.9.0
huggingface-hub huggingface-hub
lmcache==0.4.2 lmcache==0.4.5
packaging>=24.2 packaging>=24.2
typing-extensions>=4.8.0 typing-extensions>=4.8.0
pydantic pydantic
@@ -11,5 +11,5 @@ pydantic-settings
hf-transfer hf-transfer
transformers>=5 transformers>=5
bitsandbytes>=0.45.0 bitsandbytes>=0.45.0
kernels kernels<0.15
torch-c-dlpack-ext torch-c-dlpack-ext
+3
View File
@@ -97,10 +97,13 @@ If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. | | `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. | | `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. | | `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. |
| `VLLM_USE_DEEP_GEMM` | `0` | `str` (`0`/`1`) | Enable DeepGEMM FP8 kernels for MoE and MQA logits computation. Disabled by default. Must be `"0"` or `"1"` — not `true`/`false`. See note below. |
| `ATTENTION_BACKEND` | `None` | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `FLASHINFER`, `TRITON_FLASH_ATTN`). Replaces deprecated `VLLM_ATTENTION_BACKEND`. | | `ATTENTION_BACKEND` | `None` | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `FLASHINFER`, `TRITON_FLASH_ATTN`). Replaces deprecated `VLLM_ATTENTION_BACKEND`. |
| `ASYNC_SCHEDULING` | `None` | `bool` | Enable async scheduling (overlaps engine scheduling with GPU execution). Default: enabled in vLLM 0.14.0+. Set to `false` to disable. | | `ASYNC_SCHEDULING` | `None` | `bool` | Enable async scheduling (overlaps engine scheduling with GPU execution). Default: enabled in vLLM 0.14.0+. Set to `false` to disable. |
| `STREAM_INTERVAL` | `1` | `int` | Controls how often to yield streaming results. Lower = more frequent updates. | | `STREAM_INTERVAL` | `1` | `int` | Controls how often to yield streaming results. Lower = more frequent updates. |
> **Note (`VLLM_USE_DEEP_GEMM`):** DeepGEMM is used in two places: MoE weight computation and MQA logits computation. It is necessary for MQA logits computation on supported hardware — required for DeepSeek V4 models. Set `VLLM_USE_DEEP_GEMM=1` to enable. Set `VLLM_USE_DEEP_GEMM=0` to disable the MoE part and fall back to flashinfer/cutlass FP8 kernels. **Value must be `"0"` or `"1"` — not `"true"`/`"false"`.** Some users report better performance with `VLLM_USE_DEEP_GEMM=0`, particularly on H20 GPUs. Disabling it also skips the DeepGEMM warmup phase, reducing cold-start time. Requires CUDA 13.0+ and SM90+ (H100/H200) to use; the library is installed but inactive by default.
## Tokenizer Settings ## Tokenizer Settings
| Variable | Default | Type/Choices | Description | | Variable | Default | Type/Choices | Description |
+34 -17
View File
@@ -1,4 +1,5 @@
import asyncio import asyncio
import inspect
import json import json
import logging import logging
import os import os
@@ -7,6 +8,7 @@ from typing import AsyncGenerator, Optional
from dotenv import load_dotenv from dotenv import load_dotenv
from vllm import AsyncLLMEngine from vllm import AsyncLLMEngine
from vllm.inputs import TextPrompt
from vllm.entrypoints.logger import RequestLogger from vllm.entrypoints.logger import RequestLogger
from vllm.entrypoints.anthropic.protocol import AnthropicMessagesRequest, AnthropicMessagesResponse, AnthropicError, AnthropicErrorResponse from vllm.entrypoints.anthropic.protocol import AnthropicMessagesRequest, AnthropicMessagesResponse, AnthropicError, AnthropicErrorResponse
from vllm.entrypoints.anthropic.serving import AnthropicServingMessages from vllm.entrypoints.anthropic.serving import AnthropicServingMessages
@@ -30,20 +32,33 @@ class vLLMEngine:
def __init__(self, engine = None): def __init__(self, engine = None):
load_dotenv() # For local development load_dotenv() # For local development
self.engine_args = get_engine_args() self.engine_args = get_engine_args()
logging.info(f"Engine args: {self.engine_args}")
if engine is None:
# Initialize vLLM engine first ea = self.engine_args
self.llm = self._initialize_llm() if engine is None else engine.llm summary = {
"model": ea.model,
# Only create custom tokenizer wrapper if not using mistral tokenizer mode "dtype": ea.dtype,
# For mistral models, let vLLM handle tokenizer initialization "quantization": ea.quantization,
if self.engine_args.tokenizer_mode != 'mistral': "max_model_len": ea.max_model_len,
self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model, "tensor_parallel_size": ea.tensor_parallel_size,
self.engine_args.tokenizer_revision, "gpu_memory_utilization": ea.gpu_memory_utilization,
self.engine_args.trust_remote_code) }
if ea.tokenizer and ea.tokenizer != ea.model:
summary["tokenizer"] = ea.tokenizer
logging.info("Engine config: %s", summary)
logging.debug("Full engine args: %s", ea)
self.llm = self._initialize_llm()
if self.engine_args.tokenizer_mode != 'mistral':
self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model,
self.engine_args.tokenizer_revision,
self.engine_args.trust_remote_code)
else:
self.tokenizer = None
else: else:
# For mistral models, we'll get the tokenizer from vLLM later self.llm = engine.llm
self.tokenizer = None self.tokenizer = engine.tokenizer
self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY)) self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
self.default_batch_size = int(os.getenv("DEFAULT_BATCH_SIZE", DEFAULT_BATCH_SIZE)) self.default_batch_size = int(os.getenv("DEFAULT_BATCH_SIZE", DEFAULT_BATCH_SIZE))
@@ -116,7 +131,7 @@ class vLLMEngine:
if apply_chat_template or isinstance(llm_input, list): if apply_chat_template or isinstance(llm_input, list):
tokenizer_wrapper = self._get_tokenizer_for_chat_template() tokenizer_wrapper = self._get_tokenizer_for_chat_template()
llm_input = tokenizer_wrapper.apply_chat_template(llm_input) llm_input = tokenizer_wrapper.apply_chat_template(llm_input)
results_generator = self.llm.generate(llm_input, validated_sampling_params, request_id) results_generator = self.llm.generate(TextPrompt(prompt=llm_input), validated_sampling_params, request_id)
n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True
last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0} last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0}
@@ -285,7 +300,6 @@ class OpenAIvLLMEngine(vLLMEngine):
self.openai_serving_render = OpenAIServingRender( self.openai_serving_render = OpenAIServingRender(
model_config=self.llm.model_config, model_config=self.llm.model_config,
renderer=self.llm.renderer, renderer=self.llm.renderer,
io_processor=self.llm.io_processor,
model_registry=self.serving_models.registry, model_registry=self.serving_models.registry,
request_logger=None, request_logger=None,
chat_template=chat_template, chat_template=chat_template,
@@ -357,8 +371,11 @@ class OpenAIvLLMEngine(vLLMEngine):
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true', enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
) )
if hasattr(self.chat_engine, 'warmup'): warmup = getattr(self.chat_engine, 'warmup', None)
await self.chat_engine.warmup() if callable(warmup):
result = warmup()
if inspect.isawaitable(result):
await result
async def generate(self, openai_request: JobInput): async def generate(self, openai_request: JobInput):
# Ensure engines are ready (no-op if already initialized at startup) # Ensure engines are ready (no-op if already initialized at startup)
+4 -2
View File
@@ -1,10 +1,12 @@
from transformers import AutoTokenizer import logging
import os import os
from typing import Union from typing import Union
from transformers import AutoTokenizer
class TokenizerWrapper: class TokenizerWrapper:
def __init__(self, tokenizer_name_or_path, tokenizer_revision, trust_remote_code): def __init__(self, tokenizer_name_or_path, tokenizer_revision, trust_remote_code):
print(f"tokenizer_name_or_path: {tokenizer_name_or_path}, tokenizer_revision: {tokenizer_revision}, trust_remote_code: {trust_remote_code}") logging.debug("tokenizer_name_or_path: %s, tokenizer_revision: %s, trust_remote_code: %s", tokenizer_name_or_path, tokenizer_revision, trust_remote_code)
self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_name_or_path, revision=tokenizer_revision or "main", trust_remote_code=trust_remote_code) self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_name_or_path, revision=tokenizer_revision or "main", trust_remote_code=trust_remote_code)
self.custom_chat_template = os.getenv("CUSTOM_CHAT_TEMPLATE") self.custom_chat_template = os.getenv("CUSTOM_CHAT_TEMPLATE")
self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template) self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template)