Merge pull request #288 from runpod-workers/feat/0.20.0

feat: update to 0.20.2
This commit is contained in:
chrisvela
2026-05-26 17:17:45 -05:00
committed by GitHub
8 changed files with 64 additions and 32 deletions
+1 -1
View File
@@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
[![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) [![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.19.1) Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
--- ---
+11 -1
View File
@@ -9,7 +9,7 @@
"containerDiskInGb": 150, "containerDiskInGb": 150,
"gpuIds": "ADA_80_PRO,AMPERE_80", "gpuIds": "ADA_80_PRO,AMPERE_80",
"gpuCount": 1, "gpuCount": 1,
"allowedCudaVersions": ["12.9", "12.8"], "allowedCudaVersions": ["13.0"],
"presets": [ "presets": [
{ {
"name": "deepseek-ai/deepseek-r1-distill-llama-8b", "name": "deepseek-ai/deepseek-r1-distill-llama-8b",
@@ -805,6 +805,16 @@
"default": "expandable_segments:True", "default": "expandable_segments:True",
"advanced": true "advanced": true
} }
},
{
"key": "VLLM_USE_DEEP_GEMM",
"input": {
"name": "Use DeepGEMM",
"type": "boolean",
"description": "Enable DeepGEMM FP8 kernels (MoE and MQA logits). Disabled by default — enable on H100/H200 for potential throughput gains. Some GPUs (e.g. H20) may perform better with this off. Enabling also adds a warmup period on startup.",
"default": false,
"advanced": true
}
} }
] ]
} }
+9 -6
View File
@@ -1,16 +1,17 @@
FROM nvidia/cuda:12.9.1-base-ubuntu22.04 FROM nvidia/cuda:13.0.2-devel-ubuntu22.04
RUN apt-get update -y \ RUN apt-get update -y \
&& apt-get install -y python3-pip curl \ && apt-get install -y python3-pip curl git \
&& curl -LsSf https://astral.sh/uv/install.sh | sh && curl -LsSf https://astral.sh/uv/install.sh | sh
ENV PATH="/root/.local/bin:$PATH" ENV PATH="/root/.local/bin:$PATH"
RUN ldconfig /usr/local/cuda-12.9/compat/ RUN ldconfig /usr/local/cuda-13.0/compat/
# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels # Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
RUN uv pip install --system "packaging>=24.2" && \ RUN uv pip install --system "packaging>=24.2" && \
uv pip install --system "vllm[flashinfer]==0.19.1" --extra-index-url https://download.pytorch.org/whl/cu129 uv pip install --system "vllm[flashinfer]==0.20.2" && \
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
COPY builder/requirements.txt /requirements.txt COPY builder/requirements.txt /requirements.txt
@@ -42,7 +43,9 @@ ENV MODEL_NAME=$MODEL_NAME \
# Prevent rayon thread pool panic in containers where ulimit -u < nproc # Prevent rayon thread pool panic in containers where ulimit -u < nproc
# (tokenizers uses Rust's rayon which tries to spawn threads = CPU cores) # (tokenizers uses Rust's rayon which tries to spawn threads = CPU cores)
TOKENIZERS_PARALLELISM=false \ TOKENIZERS_PARALLELISM=false \
RAYON_NUM_THREADS=4 RAYON_NUM_THREADS=4 \
# Disable DeepGEMM MoE kernels by default; override with VLLM_USE_DEEP_GEMM=1 to enable
VLLM_USE_DEEP_GEMM=0
ENV PYTHONPATH="/:/vllm-workspace" ENV PYTHONPATH="/:/vllm-workspace"
+2 -2
View File
@@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png) ![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png)
Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.19.1) Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep) > Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
@@ -47,7 +47,7 @@ Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag
**📦 Docker Image**: `runpod/worker-v1-vllm:<version>` **📦 Docker Image**: `runpod/worker-v1-vllm:<version>`
- **Available Versions**: See [GitHub Releases](https://github.com/runpod-workers/worker-vllm/releases) - **Available Versions**: See [GitHub Releases](https://github.com/runpod-workers/worker-vllm/releases)
- **CUDA Compatibility**: Requires CUDA >= 12.1 - **CUDA Compatibility**: Requires CUDA >= 13.0
### Configuration ### Configuration
+1 -1
View File
@@ -3,7 +3,7 @@ pandas
pyarrow pyarrow
runpod==1.9.0 runpod==1.9.0
huggingface-hub huggingface-hub
lmcache==0.4.2 lmcache==0.4.5
packaging>=24.2 packaging>=24.2
typing-extensions>=4.8.0 typing-extensions>=4.8.0
pydantic pydantic
+3
View File
@@ -97,10 +97,13 @@ If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. | | `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. | | `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. | | `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. |
| `VLLM_USE_DEEP_GEMM` | `0` | `bool` | Enable DeepGEMM FP8 kernels for MoE and MQA logits computation. Disabled by default. See note below. |
| `ATTENTION_BACKEND` | `None` | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `FLASHINFER`, `TRITON_FLASH_ATTN`). Replaces deprecated `VLLM_ATTENTION_BACKEND`. | | `ATTENTION_BACKEND` | `None` | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `FLASHINFER`, `TRITON_FLASH_ATTN`). Replaces deprecated `VLLM_ATTENTION_BACKEND`. |
| `ASYNC_SCHEDULING` | `None` | `bool` | Enable async scheduling (overlaps engine scheduling with GPU execution). Default: enabled in vLLM 0.14.0+. Set to `false` to disable. | | `ASYNC_SCHEDULING` | `None` | `bool` | Enable async scheduling (overlaps engine scheduling with GPU execution). Default: enabled in vLLM 0.14.0+. Set to `false` to disable. |
| `STREAM_INTERVAL` | `1` | `int` | Controls how often to yield streaming results. Lower = more frequent updates. | | `STREAM_INTERVAL` | `1` | `int` | Controls how often to yield streaming results. Lower = more frequent updates. |
> **Note (`VLLM_USE_DEEP_GEMM`):** DeepGEMM is used in two places: MoE weight computation and MQA logits computation. It is necessary for MQA logits computation on supported hardware. Set `VLLM_USE_DEEP_GEMM=0` to disable the MoE part and fall back to flashinfer/cutlass FP8 kernels. Some users report better performance with `VLLM_USE_DEEP_GEMM=0`, particularly on H20 GPUs. Disabling it also skips the DeepGEMM warmup phase, reducing cold-start time. Requires CUDA 13.0+ and SM90+ (H100/H200) to use; the library is installed but inactive by default.
## Tokenizer Settings ## Tokenizer Settings
| Variable | Default | Type/Choices | Description | | Variable | Default | Type/Choices | Description |
+33 -19
View File
@@ -1,4 +1,5 @@
import asyncio import asyncio
import inspect
import json import json
import logging import logging
import os import os
@@ -7,6 +8,7 @@ from typing import AsyncGenerator, Optional
from dotenv import load_dotenv from dotenv import load_dotenv
from vllm import AsyncLLMEngine from vllm import AsyncLLMEngine
from vllm.inputs import TextPrompt
from vllm.entrypoints.logger import RequestLogger from vllm.entrypoints.logger import RequestLogger
from vllm.entrypoints.anthropic.protocol import AnthropicMessagesRequest, AnthropicMessagesResponse, AnthropicError, AnthropicErrorResponse from vllm.entrypoints.anthropic.protocol import AnthropicMessagesRequest, AnthropicMessagesResponse, AnthropicError, AnthropicErrorResponse
from vllm.entrypoints.anthropic.serving import AnthropicServingMessages from vllm.entrypoints.anthropic.serving import AnthropicServingMessages
@@ -30,20 +32,33 @@ class vLLMEngine:
def __init__(self, engine = None): def __init__(self, engine = None):
load_dotenv() # For local development load_dotenv() # For local development
self.engine_args = get_engine_args() self.engine_args = get_engine_args()
logging.info(f"Engine args: {self.engine_args}")
if engine is None:
# Initialize vLLM engine first ea = self.engine_args
self.llm = self._initialize_llm() if engine is None else engine.llm summary = {
"model": ea.model,
# Only create custom tokenizer wrapper if not using mistral tokenizer mode "dtype": ea.dtype,
# For mistral models, let vLLM handle tokenizer initialization "quantization": ea.quantization,
if self.engine_args.tokenizer_mode != 'mistral': "max_model_len": ea.max_model_len,
self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model, "tensor_parallel_size": ea.tensor_parallel_size,
self.engine_args.tokenizer_revision, "gpu_memory_utilization": ea.gpu_memory_utilization,
self.engine_args.trust_remote_code) }
if ea.tokenizer and ea.tokenizer != ea.model:
summary["tokenizer"] = ea.tokenizer
logging.info("Engine config: %s", summary)
logging.debug("Full engine args: %s", ea)
self.llm = self._initialize_llm()
if self.engine_args.tokenizer_mode != 'mistral':
self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model,
self.engine_args.tokenizer_revision,
self.engine_args.trust_remote_code)
else:
self.tokenizer = None
else: else:
# For mistral models, we'll get the tokenizer from vLLM later self.llm = engine.llm
self.tokenizer = None self.tokenizer = engine.tokenizer
self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY)) self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
self.default_batch_size = int(os.getenv("DEFAULT_BATCH_SIZE", DEFAULT_BATCH_SIZE)) self.default_batch_size = int(os.getenv("DEFAULT_BATCH_SIZE", DEFAULT_BATCH_SIZE))
@@ -116,7 +131,7 @@ class vLLMEngine:
if apply_chat_template or isinstance(llm_input, list): if apply_chat_template or isinstance(llm_input, list):
tokenizer_wrapper = self._get_tokenizer_for_chat_template() tokenizer_wrapper = self._get_tokenizer_for_chat_template()
llm_input = tokenizer_wrapper.apply_chat_template(llm_input) llm_input = tokenizer_wrapper.apply_chat_template(llm_input)
results_generator = self.llm.generate(llm_input, validated_sampling_params, request_id) results_generator = self.llm.generate(TextPrompt(prompt=llm_input), validated_sampling_params, request_id)
n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True
last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0} last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0}
@@ -285,7 +300,6 @@ class OpenAIvLLMEngine(vLLMEngine):
self.openai_serving_render = OpenAIServingRender( self.openai_serving_render = OpenAIServingRender(
model_config=self.llm.model_config, model_config=self.llm.model_config,
renderer=self.llm.renderer, renderer=self.llm.renderer,
io_processor=self.llm.io_processor,
model_registry=self.serving_models.registry, model_registry=self.serving_models.registry,
request_logger=None, request_logger=None,
chat_template=chat_template, chat_template=chat_template,
@@ -357,10 +371,10 @@ class OpenAIvLLMEngine(vLLMEngine):
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true', enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
) )
if hasattr(self.chat_engine, 'warmup'): warmup = getattr(self.chat_engine, 'warmup', None)
import asyncio if callable(warmup):
result = self.chat_engine.warmup() result = warmup()
if asyncio.iscoroutine(result): if inspect.isawaitable(result):
await result await result
async def generate(self, openai_request: JobInput): async def generate(self, openai_request: JobInput):
+4 -2
View File
@@ -1,10 +1,12 @@
from transformers import AutoTokenizer import logging
import os import os
from typing import Union from typing import Union
from transformers import AutoTokenizer
class TokenizerWrapper: class TokenizerWrapper:
def __init__(self, tokenizer_name_or_path, tokenizer_revision, trust_remote_code): def __init__(self, tokenizer_name_or_path, tokenizer_revision, trust_remote_code):
print(f"tokenizer_name_or_path: {tokenizer_name_or_path}, tokenizer_revision: {tokenizer_revision}, trust_remote_code: {trust_remote_code}") logging.debug("tokenizer_name_or_path: %s, tokenizer_revision: %s, trust_remote_code: %s", tokenizer_name_or_path, tokenizer_revision, trust_remote_code)
self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_name_or_path, revision=tokenizer_revision or "main", trust_remote_code=trust_remote_code) self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_name_or_path, revision=tokenizer_revision or "main", trust_remote_code=trust_remote_code)
self.custom_chat_template = os.getenv("CUSTOM_CHAT_TEMPLATE") self.custom_chat_template = os.getenv("CUSTOM_CHAT_TEMPLATE")
self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template) self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template)