Compare commits

..
17 Commits
Author SHA1 Message Date
Tim PietruskyandGitHub 14b74a4989 chore: re-enable .runpod/tests.json hub tests (#295)
Release / release (push) Waiting to run
Rename tests_json back to tests.json to re-enable the automated hub
tests that were temporarily disabled in #253.

Refs: DR-1161
2026-06-01 17:52:23 +02:00
chrisvelaandGitHub 50aba8fb57 Merge pull request #293 from runpod-workers/fix/update-deep-gemm-hub-value
Release / release (push) Waiting to run
fix: update VLLM_USE_DEEP_GEMM hub to default to 0
2026-05-27 11:51:48 -05:00
velaraptor-runpod 9edc5715ce fix: update configuration.md 2026-05-27 11:33:37 -05:00
velaraptor-runpod 4c91f2c5b5 fix: update VLLM_USE_DEEP_GEMM hub to default to 0 2026-05-27 11:26:52 -05:00
chrisvelaandGitHub 6265b99348 Merge pull request #288 from runpod-workers/feat/0.20.0
feat: update to 0.20.2
2026-05-26 17:17:45 -05:00
velaraptor-runpod 026f8d700b fix: specify deepgemm commit version 2026-05-21 18:46:54 -05:00
chrisvelaandGitHub 146bdb0252 Merge branch 'main' into feat/0.20.0 2026-05-21 16:04:39 -05:00
velaraptor-runpod da01193a3d chore: fix readme 2026-05-21 15:57:40 -05:00
velaraptor-runpod c2e6cc9f61 chore: update readme with correct vllm version 2026-05-21 15:39:19 -05:00
velaraptor-runpod 69968a6b39 chore: fix logging, warning for text prompt 2026-05-21 15:34:09 -05:00
velaraptor-runpod 32b29d4c6c fix: add deepgemm, update base image and hub for cuda 13.0 2026-05-20 17:43:42 -05:00
velaraptor-runpod dcea4fc4f9 fix dockerfile 2026-05-15 12:08:31 -04:00
velaraptor-runpod 9c139e8ceb update: update to 0.20.1, update dockerfile to cuda 13 2026-05-15 11:35:53 -04:00
velaraptor-runpod 678bb4be8f feat: update to 0.20.1 for patch fixes 2026-05-07 11:58:22 -05:00
velaraptor-runpod ed315a175e merge main 2026-05-01 15:28:05 -05:00
velaraptor-runpod 747cdf5891 chore: update readme vllm version 2026-04-30 20:26:23 -05:00
velaraptor-runpodandClaude Sonnet 4.6 22356ee2b3 feat: upgrade vLLM to 0.20.0
- Bump vllm[flashinfer] to 0.20.0 in Dockerfile
- Remove io_processor param from OpenAIServingRender (dropped in 0.20.0)

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-04-30 18:39:07 -05:00
9 changed files with 64 additions and 32 deletions
+1 -1
View File
@@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
[![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.19.1)
Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
---
+11 -1
View File
@@ -9,7 +9,7 @@
"containerDiskInGb": 150,
"gpuIds": "ADA_80_PRO,AMPERE_80",
"gpuCount": 1,
"allowedCudaVersions": ["12.9", "12.8"],
"allowedCudaVersions": ["13.0"],
"presets": [
{
"name": "deepseek-ai/deepseek-r1-distill-llama-8b",
@@ -805,6 +805,16 @@
"default": "expandable_segments:True",
"advanced": true
}
},
{
"key": "VLLM_USE_DEEP_GEMM",
"input": {
"name": "Use DeepGEMM",
"type": "string",
"description": "Enable DeepGEMM FP8 kernels (MoE and MQA logits). Set to 1 to enable, 0 to disable. Required for DeepSeek V4 models. Disabled by default — enable on H100/H200 for potential throughput gains. Some GPUs (e.g. H20) may perform better with this off.",
"default": "0",
"advanced": true
}
}
]
}
+9 -6
View File
@@ -1,16 +1,17 @@
FROM nvidia/cuda:12.9.1-base-ubuntu22.04
FROM nvidia/cuda:13.0.2-devel-ubuntu22.04
RUN apt-get update -y \
&& apt-get install -y python3-pip curl \
&& apt-get install -y python3-pip curl git \
&& curl -LsSf https://astral.sh/uv/install.sh | sh
ENV PATH="/root/.local/bin:$PATH"
RUN ldconfig /usr/local/cuda-12.9/compat/
RUN ldconfig /usr/local/cuda-13.0/compat/
# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels
# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
RUN uv pip install --system "packaging>=24.2" && \
uv pip install --system "vllm[flashinfer]==0.19.1" --extra-index-url https://download.pytorch.org/whl/cu129
uv pip install --system "vllm[flashinfer]==0.20.2" && \
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
COPY builder/requirements.txt /requirements.txt
@@ -42,7 +43,9 @@ ENV MODEL_NAME=$MODEL_NAME \
# Prevent rayon thread pool panic in containers where ulimit -u < nproc
# (tokenizers uses Rust's rayon which tries to spawn threads = CPU cores)
TOKENIZERS_PARALLELISM=false \
RAYON_NUM_THREADS=4
RAYON_NUM_THREADS=4 \
# Disable DeepGEMM MoE kernels by default; override with VLLM_USE_DEEP_GEMM=1 to enable
VLLM_USE_DEEP_GEMM=0
ENV PYTHONPATH="/:/vllm-workspace"
+2 -2
View File
@@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png)
Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.19.1)
Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
@@ -47,7 +47,7 @@ Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag
**📦 Docker Image**: `runpod/worker-v1-vllm:<version>`
- **Available Versions**: See [GitHub Releases](https://github.com/runpod-workers/worker-vllm/releases)
- **CUDA Compatibility**: Requires CUDA >= 12.1
- **CUDA Compatibility**: Requires CUDA >= 13.0
### Configuration
+1 -1
View File
@@ -3,7 +3,7 @@ pandas
pyarrow
runpod==1.9.0
huggingface-hub
lmcache==0.4.2
lmcache==0.4.5
packaging>=24.2
typing-extensions>=4.8.0
pydantic
+3
View File
@@ -97,10 +97,13 @@ If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. |
| `VLLM_USE_DEEP_GEMM` | `0` | `str` (`0`/`1`) | Enable DeepGEMM FP8 kernels for MoE and MQA logits computation. Disabled by default. Must be `"0"` or `"1"` — not `true`/`false`. See note below. |
| `ATTENTION_BACKEND` | `None` | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `FLASHINFER`, `TRITON_FLASH_ATTN`). Replaces deprecated `VLLM_ATTENTION_BACKEND`. |
| `ASYNC_SCHEDULING` | `None` | `bool` | Enable async scheduling (overlaps engine scheduling with GPU execution). Default: enabled in vLLM 0.14.0+. Set to `false` to disable. |
| `STREAM_INTERVAL` | `1` | `int` | Controls how often to yield streaming results. Lower = more frequent updates. |
> **Note (`VLLM_USE_DEEP_GEMM`):** DeepGEMM is used in two places: MoE weight computation and MQA logits computation. It is necessary for MQA logits computation on supported hardware — required for DeepSeek V4 models. Set `VLLM_USE_DEEP_GEMM=1` to enable. Set `VLLM_USE_DEEP_GEMM=0` to disable the MoE part and fall back to flashinfer/cutlass FP8 kernels. **Value must be `"0"` or `"1"` — not `"true"`/`"false"`.** Some users report better performance with `VLLM_USE_DEEP_GEMM=0`, particularly on H20 GPUs. Disabling it also skips the DeepGEMM warmup phase, reducing cold-start time. Requires CUDA 13.0+ and SM90+ (H100/H200) to use; the library is installed but inactive by default.
## Tokenizer Settings
| Variable | Default | Type/Choices | Description |
+33 -19
View File
@@ -1,4 +1,5 @@
import asyncio
import inspect
import json
import logging
import os
@@ -7,6 +8,7 @@ from typing import AsyncGenerator, Optional
from dotenv import load_dotenv
from vllm import AsyncLLMEngine
from vllm.inputs import TextPrompt
from vllm.entrypoints.logger import RequestLogger
from vllm.entrypoints.anthropic.protocol import AnthropicMessagesRequest, AnthropicMessagesResponse, AnthropicError, AnthropicErrorResponse
from vllm.entrypoints.anthropic.serving import AnthropicServingMessages
@@ -30,20 +32,33 @@ class vLLMEngine:
def __init__(self, engine = None):
load_dotenv() # For local development
self.engine_args = get_engine_args()
logging.info(f"Engine args: {self.engine_args}")
# Initialize vLLM engine first
self.llm = self._initialize_llm() if engine is None else engine.llm
# Only create custom tokenizer wrapper if not using mistral tokenizer mode
# For mistral models, let vLLM handle tokenizer initialization
if self.engine_args.tokenizer_mode != 'mistral':
self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model,
self.engine_args.tokenizer_revision,
self.engine_args.trust_remote_code)
if engine is None:
ea = self.engine_args
summary = {
"model": ea.model,
"dtype": ea.dtype,
"quantization": ea.quantization,
"max_model_len": ea.max_model_len,
"tensor_parallel_size": ea.tensor_parallel_size,
"gpu_memory_utilization": ea.gpu_memory_utilization,
}
if ea.tokenizer and ea.tokenizer != ea.model:
summary["tokenizer"] = ea.tokenizer
logging.info("Engine config: %s", summary)
logging.debug("Full engine args: %s", ea)
self.llm = self._initialize_llm()
if self.engine_args.tokenizer_mode != 'mistral':
self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model,
self.engine_args.tokenizer_revision,
self.engine_args.trust_remote_code)
else:
self.tokenizer = None
else:
# For mistral models, we'll get the tokenizer from vLLM later
self.tokenizer = None
self.llm = engine.llm
self.tokenizer = engine.tokenizer
self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
self.default_batch_size = int(os.getenv("DEFAULT_BATCH_SIZE", DEFAULT_BATCH_SIZE))
@@ -116,7 +131,7 @@ class vLLMEngine:
if apply_chat_template or isinstance(llm_input, list):
tokenizer_wrapper = self._get_tokenizer_for_chat_template()
llm_input = tokenizer_wrapper.apply_chat_template(llm_input)
results_generator = self.llm.generate(llm_input, validated_sampling_params, request_id)
results_generator = self.llm.generate(TextPrompt(prompt=llm_input), validated_sampling_params, request_id)
n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True
last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0}
@@ -285,7 +300,6 @@ class OpenAIvLLMEngine(vLLMEngine):
self.openai_serving_render = OpenAIServingRender(
model_config=self.llm.model_config,
renderer=self.llm.renderer,
io_processor=self.llm.io_processor,
model_registry=self.serving_models.registry,
request_logger=None,
chat_template=chat_template,
@@ -357,10 +371,10 @@ class OpenAIvLLMEngine(vLLMEngine):
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
)
if hasattr(self.chat_engine, 'warmup'):
import asyncio
result = self.chat_engine.warmup()
if asyncio.iscoroutine(result):
warmup = getattr(self.chat_engine, 'warmup', None)
if callable(warmup):
result = warmup()
if inspect.isawaitable(result):
await result
async def generate(self, openai_request: JobInput):
+4 -2
View File
@@ -1,10 +1,12 @@
from transformers import AutoTokenizer
import logging
import os
from typing import Union
from transformers import AutoTokenizer
class TokenizerWrapper:
def __init__(self, tokenizer_name_or_path, tokenizer_revision, trust_remote_code):
print(f"tokenizer_name_or_path: {tokenizer_name_or_path}, tokenizer_revision: {tokenizer_revision}, trust_remote_code: {trust_remote_code}")
logging.debug("tokenizer_name_or_path: %s, tokenizer_revision: %s, trust_remote_code: %s", tokenizer_name_or_path, tokenizer_revision, trust_remote_code)
self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_name_or_path, revision=tokenizer_revision or "main", trust_remote_code=trust_remote_code)
self.custom_chat_template = os.getenv("CUSTOM_CHAT_TEMPLATE")
self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template)