Compare commits

...
15 Commits
Author SHA1 Message Date
chrisvelaandGitHub 1606cff557 Merge pull request #265 from runpod-workers/fix/zero-max-model-num_batches
Release / release (push) Waiting to run
fix: check for zero param and set to None
2026-02-13 15:26:06 -06:00
velaraptor-runpod e705c9494b fix: check for zero param and set to None 2026-02-13 15:23:54 -06:00
chrisvelaandGitHub b749aa5718 Merge pull request #264 from runpod-workers/fix/max_num_batched_tokens
Release / release (push) Waiting to run
fix: max num batched tokens
2026-02-13 12:38:01 -06:00
velaraptor-runpod 4705ba8a7c fix: check max_num_batched_tokenz if max_model_len not set 2026-02-13 03:29:52 -06:00
velaraptor-runpod 767c66c301 make minimal changes 2026-02-13 03:23:44 -06:00
velaraptor-runpod fefdbe21a9 update changes 2026-02-13 03:16:43 -06:00
velaraptor-runpod ee961ad28d Update hub.json 2026-02-13 03:08:19 -06:00
velaraptor-runpod 2e8c251447 Merge branch 'main' into feat/update-vllm-v0.15.0 2026-02-13 03:01:05 -06:00
velaraptor-runpod c3cf43b228 Update hub.json 2026-02-13 00:22:16 -06:00
velaraptor-runpod 7ec10b98cd Update utils.py 2026-02-12 15:28:31 -06:00
c45ac42acd vLLM Worker v0.15.0 — Upgrade from v0.11.x to v0.15.0 (#259)
Release / release (push) Waiting to run
* VLLM upgrade to 0.12.0 and compatibility fixes

* MAX_NUM_BATCHED_TOKENS fix and CUDA tester

* Sys kill worker instead of marking as failed

* upgrade to vllm 0.12.0

* Update to vllm 0.15.0 and lora fix

* Update for HUB and removal of deprected env variables

* reverted docker-bake changes

* removed leftovers

* Update src/handler.py

Co-authored-by: Dj Isaac <contact@dejaydev.com>

* Update src/utils.py

Co-authored-by: Dj Isaac <contact@dejaydev.com>

* Update src/handler.py

Co-authored-by: Dj Isaac <contact@dejaydev.com>

* Clean up of docs and comments in code

* nit: lowercase p

* nit: lowercase p

---------

Co-authored-by: Dj Isaac <contact@dejaydev.com>
Co-authored-by: chrisvela <chris.vela@runpod.io>
2026-02-12 21:50:34 +01:00
velaraptor-runpod 340bc0b3c6 fix: served model name 2026-02-10 21:42:58 -06:00
velaraptor-runpod e1e9ef74ad add changes from pr 2026-02-06 18:10:09 -06:00
velaraptor-runpod 461f89cea6 add torch-c-dlpack-ext requirement 2026-02-06 17:03:39 -06:00
velaraptor-runpod 8eb55b90c1 add changes for v0.15.0 2026-02-05 17:24:16 -06:00
8 changed files with 388 additions and 348 deletions
+36 -262
View File
@@ -9,7 +9,7 @@
"containerDiskInGb": 150,
"gpuIds": "ADA_80_PRO,AMPERE_80",
"gpuCount": 1,
"allowedCudaVersions": ["12.9", "12.8", "12.7", "12.6", "12.5", "12.4"],
"allowedCudaVersions": ["12.9", "12.8"],
"presets": [
{
"name": "deepseek-ai/deepseek-r1-distill-llama-8b",
@@ -181,41 +181,13 @@
"advanced": true
}
},
{
"key": "QUANTIZATION_PARAM_PATH",
"input": {
"name": "Quantization Param Path",
"type": "string",
"description": "Path to the JSON file containing the KV cache scaling factors.",
"advanced": true
}
},
{
"key": "MAX_MODEL_LEN",
"input": {
"name": "Max Model Length",
"type": "number",
"description": "Model context length.",
"advanced": true
}
},
{
"key": "GUIDED_DECODING_BACKEND",
"input": {
"name": "Guided Decoding Backend",
"type": "string",
"description": "Which engine will be used for guided decoding by default.",
"options": [
{
"label": "outlines",
"value": "outlines"
},
{
"label": "lm-format-enforcer",
"value": "lm-format-enforcer"
}
],
"default": "outlines",
"default": null,
"advanced": true
}
},
@@ -235,17 +207,8 @@
"value": "mp"
}
],
"advanced": true
}
},
{
"key": "WORKER_USE_RAY",
"input": {
"name": "Worker Use Ray",
"type": "boolean",
"description": "Deprecated, use --distributed-executor-backend=ray.",
"default": false,
"advanced": true
"advanced": true,
"default": "mp"
}
},
{
@@ -307,26 +270,6 @@
"advanced": true
}
},
{
"key": "USE_V2_BLOCK_MANAGER",
"input": {
"name": "Use V2 Block Manager",
"type": "boolean",
"description": "Use BlockSpaceMangerV2.",
"default": false,
"advanced": true
}
},
{
"key": "NUM_LOOKAHEAD_SLOTS",
"input": {
"name": "Num Lookahead Slots",
"type": "number",
"description": "Experimental scheduling config necessary for speculative decoding.",
"default": 0,
"advanced": true
}
},
{
"key": "SEED",
"input": {
@@ -352,6 +295,7 @@
"name": "Max Num Batched Tokens",
"type": "number",
"description": "Maximum number of batched tokens per iteration.",
"default": null,
"advanced": true
}
},
@@ -412,53 +356,6 @@
"advanced": true
}
},
{
"key": "ROPE_SCALING",
"input": {
"name": "RoPE Scaling",
"type": "string",
"description": "RoPE scaling configuration in JSON format.",
"advanced": true
}
},
{
"key": "ROPE_THETA",
"input": {
"name": "RoPE Theta",
"type": "number",
"description": "RoPE theta. Use with rope_scaling.",
"advanced": true
}
},
{
"key": "TOKENIZER_POOL_SIZE",
"input": {
"name": "Tokenizer Pool Size",
"type": "number",
"description": "Size of tokenizer pool to use for asynchronous tokenization.",
"default": 0,
"advanced": true
}
},
{
"key": "TOKENIZER_POOL_TYPE",
"input": {
"name": "Tokenizer Pool Type",
"type": "string",
"description": "Type of tokenizer pool to use for asynchronous tokenization.",
"default": "ray",
"advanced": true
}
},
{
"key": "TOKENIZER_POOL_EXTRA_CONFIG",
"input": {
"name": "Tokenizer Pool Extra Config",
"type": "string",
"description": "Extra config for tokenizer pool.",
"advanced": true
}
},
{
"key": "ENABLE_LORA",
"input": {
@@ -489,16 +386,6 @@
"advanced": true
}
},
{
"key": "LORA_EXTRA_VOCAB_SIZE",
"input": {
"name": "LoRA Extra Vocab Size",
"type": "number",
"description": "Maximum size of extra vocabulary for LoRA adapters.",
"default": 256,
"advanced": true
}
},
{
"key": "LORA_DTYPE",
"input": {
@@ -527,15 +414,6 @@
"advanced": true
}
},
{
"key": "LONG_LORA_SCALING_FACTORS",
"input": {
"name": "Long LoRA Scaling Factors",
"type": "string",
"description": "Specify multiple scaling factors for LoRA adapters.",
"advanced": true
}
},
{
"key": "MAX_CPU_LORAS",
"input": {
@@ -615,6 +493,34 @@
"advanced": true
}
},
{
"key": "SPECULATIVE_CONFIG",
"input": {
"name": "Speculative Config (JSON)",
"type": "string",
"description": "Full speculative decoding configuration as a JSON string. Overrides individual speculative env vars.",
"advanced": true
}
},
{
"key": "SPECULATIVE_METHOD",
"input": {
"name": "Speculative Method",
"type": "string",
"description": "Speculative decoding method to use.",
"options": [
{ "label": "None", "value": "" },
{ "label": "Draft Model", "value": "draft_model" },
{ "label": "N-gram", "value": "ngram" },
{ "label": "EAGLE", "value": "eagle" },
{ "label": "EAGLE3", "value": "eagle3" },
{ "label": "Medusa", "value": "medusa" },
{ "label": "MLP Speculator", "value": "mlp_speculator" }
],
"default": "",
"advanced": true
}
},
{
"key": "SPECULATIVE_MODEL",
"input": {
@@ -633,33 +539,6 @@
"advanced": true
}
},
{
"key": "SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE",
"input": {
"name": "Speculative Draft Tensor Parallel Size",
"type": "number",
"description": "Number of tensor parallel replicas for the draft model.",
"advanced": true
}
},
{
"key": "SPECULATIVE_MAX_MODEL_LEN",
"input": {
"name": "Speculative Max Model Length",
"type": "number",
"description": "The maximum sequence length supported by the draft model.",
"advanced": true
}
},
{
"key": "SPECULATIVE_DISABLE_BY_BATCH_SIZE",
"input": {
"name": "Speculative Disable by Batch Size",
"type": "number",
"description": "Disable speculative decoding if the number of enqueue requests is larger than this value.",
"advanced": true
}
},
{
"key": "NGRAM_PROMPT_LOOKUP_MAX",
"input": {
@@ -669,53 +548,6 @@
"advanced": true
}
},
{
"key": "NGRAM_PROMPT_LOOKUP_MIN",
"input": {
"name": "Ngram Prompt Lookup Min",
"type": "number",
"description": "Min size of window for ngram prompt lookup in speculative decoding.",
"advanced": true
}
},
{
"key": "SPEC_DECODING_ACCEPTANCE_METHOD",
"input": {
"name": "Speculative Decoding Acceptance Method",
"type": "string",
"description": "Specify the acceptance method for draft token verification in speculative decoding.",
"options": [
{
"label": "rejection_sampler",
"value": "rejection_sampler"
},
{
"label": "typical_acceptance_sampler",
"value": "typical_acceptance_sampler"
}
],
"default": "rejection_sampler",
"advanced": true
}
},
{
"key": "TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD",
"input": {
"name": "Typical Acceptance Sampler Posterior Threshold",
"type": "number",
"description": "Set the lower bound threshold for the posterior probability of a token to be accepted.",
"advanced": true
}
},
{
"key": "TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA",
"input": {
"name": "Typical Acceptance Sampler Posterior Alpha",
"type": "number",
"description": "A scaling factor for the entropy-based threshold for token acceptance.",
"advanced": true
}
},
{
"key": "MODEL_LOADER_EXTRA_CONFIG",
"input": {
@@ -726,49 +558,11 @@
}
},
{
"key": "PREEMPTION_MODE",
"key": "ENABLE_LOG_REQUESTS",
"input": {
"name": "Preemption Mode",
"type": "string",
"description": "If 'recompute', the engine performs preemption-aware recomputation. If 'save', the engine saves activations into the CPU memory as preemption happens.",
"advanced": true
}
},
{
"key": "PREEMPTION_CHECK_PERIOD",
"input": {
"name": "Preemption Check Period",
"type": "number",
"description": "How frequently the engine checks if a preemption happens.",
"default": 1,
"advanced": true
}
},
{
"key": "PREEMPTION_CPU_CAPACITY",
"input": {
"name": "Preemption CPU Capacity",
"type": "number",
"description": "The percentage of CPU memory used for the saved activations.",
"default": 2,
"advanced": true
}
},
{
"key": "MAX_LOG_LEN",
"input": {
"name": "Max Log Length",
"type": "number",
"description": "Max number of characters or ID numbers being printed in log.",
"advanced": true
}
},
{
"key": "DISABLE_LOGGING_REQUEST",
"input": {
"name": "Disable Logging Request",
"name": "Enable Log Requests",
"type": "boolean",
"description": "Disable logging requests.",
"description": "Enable vLLM request logging.",
"default": false,
"advanced": true
}
@@ -840,16 +634,6 @@
"advanced": true
}
},
{
"key": "MAX_SEQ_LEN_TO_CAPTURE",
"input": {
"name": "CUDA Graph Max Content Length",
"type": "number",
"description": "Maximum context length covered by CUDA graphs. If a sequence has context length larger than this, we fall back to eager mode",
"default": 8192,
"advanced": true
}
},
{
"key": "DISABLE_CUSTOM_ALL_REDUCE",
"input": {
@@ -958,16 +742,6 @@
"advanced": true
}
},
{
"key": "DISABLE_LOG_REQUESTS",
"input": {
"name": "Disable Log Requests",
"type": "boolean",
"description": "Enables or disables vLLM request logging",
"default": true,
"advanced": true
}
},
{
"key": "ENABLE_AUTO_TOOL_CHOICE",
"input": {
+23 -8
View File
@@ -1,19 +1,21 @@
FROM nvidia/cuda:12.4.1-base-ubuntu22.04
FROM nvidia/cuda:12.8.0-base-ubuntu22.04
RUN apt-get update -y \
&& apt-get install -y python3-pip
RUN ldconfig /usr/local/cuda-12.4/compat/
RUN ldconfig /usr/local/cuda-12.8/compat/
# Install Python dependencies
# Install vLLM with FlashInfer - use CUDA 12.8 PyTorch wheels (compatible with vLLM 0.15.0)
RUN python3 -m pip install --upgrade pip && \
python3 -m pip install "vllm[flashinfer]==0.15.0" --extra-index-url https://download.pytorch.org/whl/cu128
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
COPY builder/requirements.txt /requirements.txt
RUN --mount=type=cache,target=/root/.cache/pip \
python3 -m pip install --upgrade pip && \
python3 -m pip install --upgrade -r /requirements.txt
# Install vLLM
RUN python3 -m pip install vllm==0.11.0
# Setup for Option 2: Building the Image with the Model included
ARG MODEL_NAME=""
ARG TOKENIZER_NAME=""
@@ -21,6 +23,7 @@ ARG BASE_PATH="/runpod-volume"
ARG QUANTIZATION=""
ARG MODEL_REVISION=""
ARG TOKENIZER_REVISION=""
ARG VLLM_NIGHTLY="false"
ENV MODEL_NAME=$MODEL_NAME \
MODEL_REVISION=$MODEL_REVISION \
@@ -31,10 +34,22 @@ ENV MODEL_NAME=$MODEL_NAME \
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
HUGGINGFACE_HUB_CACHE="${BASE_PATH}/huggingface-cache/hub" \
HF_HOME="${BASE_PATH}/huggingface-cache/hub" \
HF_HUB_ENABLE_HF_TRANSFER=0
HF_HUB_ENABLE_HF_TRANSFER=0 \
# Suppress Ray metrics agent warnings (not needed in containerized environments)
RAY_METRICS_EXPORT_ENABLED=0 \
RAY_DISABLE_USAGE_STATS=1 \
# Prevent rayon thread pool panic in containers where ulimit -u < nproc
# (tokenizers uses Rust's rayon which tries to spawn threads = CPU cores)
TOKENIZERS_PARALLELISM=false \
RAYON_NUM_THREADS=4
ENV PYTHONPATH="/:/vllm-workspace"
RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \
pip install -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \
apt-get update && apt-get install -y git && rm -rf /var/lib/apt/lists/* && \
pip install git+https://github.com/huggingface/transformers.git; \
fi
COPY src /src
RUN --mount=type=secret,id=HF_TOKEN,required=false \
+2 -2
View File
@@ -1,7 +1,7 @@
ray
pandas
pyarrow
runpod>=1.8,<2.0
runpod
huggingface-hub
packaging
typing-extensions>=4.8.0
@@ -11,4 +11,4 @@ hf-transfer
transformers>=4.57.0
bitsandbytes>=0.45.0
kernels
torch==2.6.0
torch-c-dlpack-ext
+36 -11
View File
@@ -28,7 +28,6 @@ Complete guide to all environment variables and configuration options for worker
| `RAY_WORKERS_USE_NSIGHT` | False | `bool` | If specified, use nsight to profile Ray workers. |
| `ENABLE_PREFIX_CACHING` | False | `bool` | Enables automatic prefix caching. |
| `DISABLE_SLIDING_WINDOW` | False | `bool` | Disables sliding window, capping to sliding window size. |
| `USE_V2_BLOCK_MANAGER` | False | `bool` | Use BlockSpaceMangerV2. |
| `NUM_LOOKAHEAD_SLOTS` | 0 | `int` | Experimental scheduling config necessary for speculative decoding. |
| `SEED` | 0 | `int` | Random seed for operations. |
| `NUM_GPU_BLOCKS_OVERRIDE` | None | `int` | If specified, ignore GPU profiling result and use this number of GPU blocks. |
@@ -57,12 +56,25 @@ Complete guide to all environment variables and configuration options for worker
| `FULLY_SHARDED_LORAS` | False | `bool` | Enable fully sharded LoRA layers. |
| `LORA_MODULES` | `[]` | `list[dict]` | Add lora adapters from Hugging Face `[{"name": "xx", "path": "xxx/xxxx", "base_model_name": "xxx/xxxx"}]` |
> **Note (Serverless)**: When LoRA adapters are configured via `LORA_MODULES`, initialization is deferred to the first request to ensure compatibility with RunPod Serverless. This means the first request will include LoRA loading time. Subsequent requests are unaffected. Check logs for "LoRA mode: X adapter(s) will load on first request" at startup.
## Speculative Decoding Settings
Speculative decoding can be configured in two ways:
### Option 1: JSON Configuration
Set `SPECULATIVE_CONFIG` to a JSON string with your full speculative decoding configuration:
```bash
SPECULATIVE_CONFIG='{"method": "ngram", "num_speculative_tokens": 5, "prompt_lookup_max": 4}'
```
### Option 2: Individual Environment Variables
| Variable | Default | Type/Choices | Description |
| ------------------------------------------------ | ------------------- | --------------------------------------------------- | ----------------------------------------------------------------------------------------- |
| `SCHEDULER_DELAY_FACTOR` | 0.0 | `float` | Apply a delay before scheduling next prompt. |
| `ENABLE_CHUNKED_PREFILL` | False | `bool` | Enable chunked prefill requests. |
| ---------------------------------------- | ------- | ------------------------------------------------------------------ | ----------------------------------------------------------------------------------------- |
| `SPECULATIVE_METHOD` | None | ['draft_model', 'ngram', 'eagle', 'eagle3', 'medusa', 'mlp_speculator'] | Speculative decoding method to use. |
| `SPECULATIVE_MODEL` | None | `str` | The name of the draft model to be used in speculative decoding. |
| `NUM_SPECULATIVE_TOKENS` | None | `int` | The number of speculative tokens to sample from the draft model. |
| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | None | `int` | Number of tensor parallel replicas for the draft model. |
@@ -70,11 +82,10 @@ Complete guide to all environment variables and configuration options for worker
| `SPECULATIVE_DISABLE_BY_BATCH_SIZE` | None | `int` | Disable speculative decoding if the number of enqueue requests is larger than this value. |
| `NGRAM_PROMPT_LOOKUP_MAX` | None | `int` | Max size of window for ngram prompt lookup in speculative decoding. |
| `NGRAM_PROMPT_LOOKUP_MIN` | None | `int` | Min size of window for ngram prompt lookup in speculative decoding. |
| `SPEC_DECODING_ACCEPTANCE_METHOD` | 'rejection_sampler' | ['rejection_sampler', 'typical_acceptance_sampler'] | Specify the acceptance method for draft token verification in speculative decoding. |
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD` | None | `float` | Set the lower bound threshold for the posterior probability of a token to be accepted. |
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA` | None | `float` | A scaling factor for the entropy-based threshold for token acceptance. |
## System Performance Settings
If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When using individual env vars without `SPECULATIVE_METHOD`, the method is auto-detected from the model name or configuration.
## Scheduling & Performance Settings
| Variable | Default | Type/Choices | Description |
| ------------------------------ | ------- | --------------- | ----------------------------------------------------------------------------------------------------------------------------------- |
@@ -85,7 +96,10 @@ Complete guide to all environment variables and configuration options for worker
| `ENFORCE_EAGER` | False | `bool` | Always use eager-mode PyTorch. If False(`0`), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility. |
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models |
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. |
| `ATTENTION_BACKEND` | `None` | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `FLASHINFER`, `TRITON_FLASH_ATTN`). Replaces deprecated `VLLM_ATTENTION_BACKEND`. |
| `ASYNC_SCHEDULING` | `None` | `bool` | Enable async scheduling (overlaps engine scheduling with GPU execution). Default: enabled in vLLM 0.14.0+. Set to `false` to disable. |
| `STREAM_INTERVAL` | `1` | `int` | Controls how often to yield streaming results. Lower = more frequent updates. |
## Tokenizer Settings
@@ -115,6 +129,13 @@ The way this works is that the first request will have a batch size of `DEFAULT_
| `ENABLE_AUTO_TOOL_CHOICE` | `false` | `bool` | Enables automatic tool selection for supported models. Set to `true` to activate. |
| `TOOL_CALL_PARSER` | `None` | `str` | Specifies the parser for tool calls. Options: `mistral`, `hermes`, `llama3_json`, `llama4_json`, `llama4_pythonic`, `granite`, `granite-20b-fc`, `deepseek_v3`, `internlm`, `jamba`, `phi4_mini_json`, `pythonic` |
| `REASONING_PARSER` | `None` | `str` | Parser for reasoning-capable models (enables reasoning mode). Examples: `deepseek_r1`, `qwen3`, `granite`, `hunyuan_a13b`. Leave unset to disable. |
| `TRUST_REQUEST_CHAT_TEMPLATE` | `false` | `bool` | Allow clients to send custom chat templates in API requests. **Security consideration:** Only enable if you trust your API clients. |
| `RETURN_TOKENS_AS_TOKEN_IDS` | `false` | `bool` | Return token IDs instead of decoded text strings in responses. |
| `EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE` | `false` | `bool` | Exclude tool definitions from the prompt when `tool_choice` is set to `none`. |
| `ENABLE_PROMPT_TOKENS_DETAILS` | `false` | `bool` | Include detailed prompt token information in API responses. |
| `ENABLE_FORCE_INCLUDE_USAGE` | `false` | `bool` | Always include usage statistics in API responses, even when not requested. |
| `ENABLE_LOG_OUTPUTS` | `false` | `bool` | Log model outputs for debugging purposes. |
| `LOG_ERROR_STACK` | `false` | `bool` | Include full stack traces in error responses for debugging. |
## Serverless & Concurrency Settings
@@ -122,7 +143,7 @@ The way this works is that the first request will have a batch size of `DEFAULT_
| ---------------------- | ------- | ------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `MAX_CONCURRENCY` | `30` | `int` | Max concurrent requests per worker. vLLM has an internal queue, so you don't have to worry about limiting by VRAM, this is for improving scaling/load balancing efficiency |
| `DISABLE_LOG_STATS` | False | `bool` | Enables or disables vLLM stats logging. |
| `DISABLE_LOG_REQUESTS` | False | `bool` | Enables or disables vLLM request logging. |
| `ENABLE_LOG_REQUESTS` | False | `bool` | Enables vLLM request logging. (Replaces deprecated `DISABLE_LOG_REQUESTS` in vLLM 0.15.0) |
## Advanced Settings
@@ -149,6 +170,10 @@ These variables are used when building custom Docker images with models baked in
⚠️ **The following variables are deprecated and will be removed in future versions:**
| Old Variable | New Variable | Note |
| ---------------------------- | ------------------------ | --------------------- |
| ---------------------------- | ------------------------ | -------------------------------------------------------------------- |
| `MAX_CONTEXT_LEN_TO_CAPTURE` | `MAX_SEQ_LEN_TO_CAPTURE` | Use new variable name |
| `kv_cache_dtype=fp8_e5m2` | `kv_cache_dtype=fp8` | Simplified fp8 format |
| `USE_V2_BLOCK_MANAGER` | *(removed)* | V2 block manager is now the default in vLLM 0.13.0, setting ignored |
| `VLLM_ATTENTION_BACKEND` | `ATTENTION_BACKEND` | Use new env var name (old still works with deprecation warning) |
| `DISABLE_LOG_REQUESTS` | `ENABLE_LOG_REQUESTS` | Inverted logic in vLLM 0.15.0 (old still works with deprecation warning) |
+65 -26
View File
@@ -1,24 +1,25 @@
import os
import logging
import json
import asyncio
import json
import logging
import os
import time
from typing import AsyncGenerator, Optional
from dotenv import load_dotenv
from typing import AsyncGenerator, Optional
import time
from vllm import AsyncLLMEngine
from vllm.entrypoints.logger import RequestLogger
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
from vllm.entrypoints.openai.serving_completion import OpenAIServingCompletion
from vllm.entrypoints.openai.protocol import ChatCompletionRequest, CompletionRequest, ErrorResponse
from vllm.entrypoints.openai.serving_models import BaseModelPath, LoRAModulePath, OpenAIServingModels
from vllm.entrypoints.openai.chat_completion.protocol import ChatCompletionRequest
from vllm.entrypoints.openai.chat_completion.serving import OpenAIServingChat
from vllm.entrypoints.openai.completion.protocol import CompletionRequest
from vllm.entrypoints.openai.completion.serving import OpenAIServingCompletion
from vllm.entrypoints.openai.engine.protocol import ErrorResponse
from vllm.entrypoints.openai.models.protocol import BaseModelPath, LoRAModulePath
from vllm.entrypoints.openai.models.serving import OpenAIServingModels
from utils import DummyRequest, JobInput, BatchSize, create_error_response
from constants import DEFAULT_MAX_CONCURRENCY, DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MIN_BATCH_SIZE
from tokenizer import TokenizerWrapper
from constants import DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MAX_CONCURRENCY, DEFAULT_MIN_BATCH_SIZE
from engine_args import get_engine_args
from tokenizer import TokenizerWrapper
from utils import BatchSize, DummyRequest, JobInput, create_error_response
class vLLMEngine:
def __init__(self, engine = None):
@@ -174,10 +175,24 @@ class vLLMEngine:
class OpenAIvLLMEngine(vLLMEngine):
def __init__(self, vllm_engine):
super().__init__(vllm_engine)
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.engine_args.model
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.engine_args.served_model_name or self.engine_args.model
self.response_role = os.getenv("OPENAI_RESPONSE_ROLE") or "assistant"
self.lora_adapters = self._load_lora_adapters()
asyncio.run(self._initialize_engines())
# Always defer OpenAI engine initialization to the first request.
# asyncio.run() creates a temporary event loop that gets closed, but async
# components (tokenizer pool, serving engines) bind futures to that loop.
# When Runpod's serverless handler runs in its own event loop, those futures
# are "attached to a different loop" causing RuntimeError.
# This affects all configurations, not just LoRA.
self._engines_initialized = False
if self.lora_adapters:
logging.info(f"LoRA mode: {len(self.lora_adapters)} adapter(s) will load on first request")
for adapter in self.lora_adapters:
logging.info(f" - {adapter.name}: {adapter.path}")
else:
logging.info("OpenAI engines will initialize on first request")
# Handle both integer and boolean string values for RAW_OPENAI_OUTPUT
raw_output_env = os.getenv("RAW_OPENAI_OUTPUT", "1")
if raw_output_env.lower() in ('true', 'false'):
@@ -201,15 +216,28 @@ class OpenAIvLLMEngine(vLLMEngine):
continue
return adapters
async def _ensure_engines_initialized(self):
"""Initialize engines on first request to avoid event loop mismatch.
In Runpod Serverless, the startup code runs outside the handler's event
loop. Deferring initialization to the first request ensures all async
components (tokenizer pool, serving engines, LoRA state) are created in
the correct event loop context.
"""
if not self._engines_initialized:
logging.info("Initializing OpenAI serving engines...")
await self._initialize_engines()
self._engines_initialized = True
logging.info("OpenAI serving engines initialized successfully")
async def _initialize_engines(self):
self.model_config = await self.llm.get_model_config()
self.model_config = self.llm.model_config
self.base_model_paths = [
BaseModelPath(name=self.engine_args.model, model_path=self.engine_args.model)
BaseModelPath(name=self.served_model_name, model_path=self.engine_args.model)
]
self.serving_models = OpenAIServingModels(
engine_client=self.llm,
model_config=self.model_config,
base_model_paths=self.base_model_paths,
lora_modules=self.lora_adapters,
)
@@ -222,28 +250,39 @@ class OpenAIvLLMEngine(vLLMEngine):
self.chat_engine = OpenAIServingChat(
engine_client=self.llm,
model_config=self.model_config,
models=self.serving_models,
response_role=self.response_role,
request_logger=None,
chat_template=chat_template,
chat_template_content_format="auto",
# enable_reasoning=os.getenv('ENABLE_REASONING', 'false').lower() == 'true',
reasoning_parser= os.getenv('REASONING_PARSER', "") or None,
# return_token_as_token_ids=False,
trust_request_chat_template=os.getenv('TRUST_REQUEST_CHAT_TEMPLATE', 'false').lower() == 'true',
return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true',
reasoning_parser=os.getenv('REASONING_PARSER', "") or "",
enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true',
exclude_tools_when_tool_choice_none=os.getenv('EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE', 'false').lower() == 'true',
tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None,
enable_prompt_tokens_details=False
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true',
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
)
self.completion_engine = OpenAIServingCompletion(
engine_client=self.llm,
model_config=self.model_config,
models=self.serving_models,
request_logger=None,
# return_token_as_token_ids=False,
return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true',
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
)
if hasattr(self.chat_engine, 'warmup'):
await self.chat_engine.warmup()
async def generate(self, openai_request: JobInput):
# Ensure engines are ready (no-op if already initialized at startup)
await self._ensure_engines_initialized()
if openai_request.openai_route == "/v1/models":
yield await self._handle_model_request()
elif openai_request.openai_route in ["/v1/chat/completions", "/v1/completions"]:
+158 -3
View File
@@ -15,7 +15,8 @@ RENAME_ARGS_MAP = {
DEFAULT_ARGS = {
"disable_log_stats": os.getenv('DISABLE_LOG_STATS', 'False').lower() == 'true',
"disable_log_requests": os.getenv('DISABLE_LOG_REQUESTS', 'False').lower() == 'true',
# disable_log_requests is deprecated, use enable_log_requests instead
"enable_log_requests": os.getenv('ENABLE_LOG_REQUESTS', 'False').lower() == 'true',
"gpu_memory_utilization": float(os.getenv('GPU_MEMORY_UTILIZATION', 0.95)),
"pipeline_parallel_size": int(os.getenv('PIPELINE_PARALLEL_SIZE', 1)),
"tensor_parallel_size": int(os.getenv('TENSOR_PARALLEL_SIZE', 1)),
@@ -38,9 +39,15 @@ DEFAULT_ARGS = {
"block_size": int(os.getenv('BLOCK_SIZE', 16)),
"enable_prefix_caching": os.getenv('ENABLE_PREFIX_CACHING', 'False').lower() == 'true',
"disable_sliding_window": os.getenv('DISABLE_SLIDING_WINDOW', 'False').lower() == 'true',
"use_v2_block_manager": os.getenv('USE_V2_BLOCK_MANAGER', 'False').lower() == 'true',
# attention_backend replaces deprecated VLLM_ATTENTION_BACKEND env var
"attention_backend": os.getenv('ATTENTION_BACKEND', None),
# Enabled by default for improved throughput. Set to False to disable if experiencing issues
"async_scheduling": None if os.getenv('ASYNC_SCHEDULING') is None else os.getenv('ASYNC_SCHEDULING', 'True').lower() == 'true',
# Controls how often to yield streaming results
"stream_interval": int(os.getenv('STREAM_INTERVAL', 1)),
"swap_space": int(os.getenv('SWAP_SPACE', 4)), # GiB
"cpu_offload_gb": int(os.getenv('CPU_OFFLOAD_GB', 0)), # GiB
# vLLM defaults None to 2048; keep 0 as None to let vLLM auto-calculate
"max_num_batched_tokens": int(os.getenv('MAX_NUM_BATCHED_TOKENS', 0)) or None,
"max_num_seqs": int(os.getenv('MAX_NUM_SEQS', 256)),
"max_logprobs": int(os.getenv('MAX_LOGPROBS', 20)), # Default value for OpenAI Chat Completions API
@@ -92,8 +99,111 @@ DEFAULT_ARGS = {
"qlora_adapter_name_or_path": os.getenv('QLORA_ADAPTER_NAME_OR_PATH', None),
"disable_logprobs_during_spec_decoding": os.getenv('DISABLE_LOGPROBS_DURING_SPEC_DECODING', None),
"otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None),
"use_v2_block_manager": os.getenv('USE_V2_BLOCK_MANAGER', 'true'),
}
def get_speculative_config():
"""Build speculative decoding configuration from environment variables.
Supports two modes:
1. Full JSON config via SPECULATIVE_CONFIG env var
2. Individual env vars for common settings
"""
# Option 1: Full JSON configuration
spec_config_json = os.getenv('SPECULATIVE_CONFIG')
if spec_config_json:
try:
config = json.loads(spec_config_json)
logging.info(f"Using speculative config from SPECULATIVE_CONFIG: {config}")
return config
except json.JSONDecodeError as e:
logging.error(f"Failed to parse SPECULATIVE_CONFIG JSON: {e}")
return None
# Option 2: Build config from individual environment variables
spec_method = os.getenv('SPECULATIVE_METHOD')
spec_model = os.getenv('SPECULATIVE_MODEL')
num_spec_tokens = os.getenv('NUM_SPECULATIVE_TOKENS')
ngram_max = os.getenv('NGRAM_PROMPT_LOOKUP_MAX')
ngram_min = os.getenv('NGRAM_PROMPT_LOOKUP_MIN')
if not any([spec_method, spec_model, ngram_max]):
return None
config = {}
# Determine method
if spec_method:
config['method'] = spec_method
elif ngram_max and not spec_model:
config['method'] = 'ngram'
elif spec_model:
model_lower = spec_model.lower()
if 'eagle3' in model_lower:
config['method'] = 'eagle3'
elif 'eagle' in model_lower:
config['method'] = 'eagle'
elif 'medusa' in model_lower:
config['method'] = 'medusa'
else:
config['method'] = 'draft_model'
if spec_model:
config['model'] = spec_model
if num_spec_tokens:
config['num_speculative_tokens'] = int(num_spec_tokens)
if ngram_max:
config['prompt_lookup_max'] = int(ngram_max)
if ngram_min:
config['prompt_lookup_min'] = int(ngram_min)
draft_tp = os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE')
if draft_tp:
config['draft_tensor_parallel_size'] = int(draft_tp)
spec_max_len = os.getenv('SPECULATIVE_MAX_MODEL_LEN')
if spec_max_len:
config['max_model_len'] = int(spec_max_len)
disable_batch = os.getenv('SPECULATIVE_DISABLE_BY_BATCH_SIZE')
if disable_batch:
config['disable_by_batch_size'] = int(disable_batch)
spec_quant = os.getenv('SPECULATIVE_QUANTIZATION')
if spec_quant:
config['quantization'] = spec_quant
spec_revision = os.getenv('SPECULATIVE_MODEL_REVISION')
if spec_revision:
config['revision'] = spec_revision
spec_eager = os.getenv('SPECULATIVE_ENFORCE_EAGER')
if spec_eager:
config['enforce_eager'] = spec_eager.lower() == 'true'
if config:
logging.info(f"Built speculative config from env vars: {config}")
return config
return None
def _resolve_max_model_len(model, trust_remote_code=False, revision=None):
"""Resolve max_model_len from the model's HuggingFace config."""
try:
from transformers import AutoConfig
config = AutoConfig.from_pretrained(
model,
trust_remote_code=trust_remote_code,
revision=revision,
)
for attr in ('max_position_embeddings', 'n_positions', 'max_seq_len', 'seq_length'):
val = getattr(config, attr, None)
if val is not None:
logging.info(f"Resolved max_model_len={val} from model config ({attr})")
return val
except Exception as e:
logging.warning(f"Could not resolve max_model_len from model config: {e}")
return None
limit_mm_env = os.getenv('LIMIT_MM_PER_PROMPT')
if limit_mm_env is not None:
DEFAULT_ARGS["limit_mm_per_prompt"] = convert_limit_mm_per_prompt(limit_mm_env)
@@ -176,4 +286,49 @@ def get_engine_args():
# os.environ["VLLM_ATTENTION_BACKEND"] = "FLASHINFER"
# logging.info("Using FLASHINFER for gemma-2 model.")
# Set max_num_batched_tokens to max_model_len for unlimited batching.
# vLLM defaults max_num_batched_tokens to 2048 when None, which is too low.
if args.get("max_model_len") == 0:
args["max_model_len"] = None
if args.get("max_num_batched_tokens") == 0:
args["max_num_batched_tokens"] = None
if args.get("max_num_batched_tokens") is None:
max_model_len = args.get("max_model_len")
if max_model_len is None:
max_model_len = _resolve_max_model_len(
args.get("model"),
trust_remote_code=args.get("trust_remote_code", False),
revision=args.get("revision"),
)
if max_model_len is not None:
args["max_num_batched_tokens"] = max_model_len
logging.info(f"Setting max_num_batched_tokens to {max_model_len}")
# VLLM_ATTENTION_BACKEND is deprecated, migrate to attention_backend
if os.getenv('VLLM_ATTENTION_BACKEND'):
logging.warning(
"VLLM_ATTENTION_BACKEND env var is deprecated. "
"Use ATTENTION_BACKEND instead (maps to --attention-backend CLI arg)."
)
if not args.get('attention_backend'):
args['attention_backend'] = os.getenv('VLLM_ATTENTION_BACKEND')
# DISABLE_LOG_REQUESTS is deprecated, use ENABLE_LOG_REQUESTS instead
if os.getenv('DISABLE_LOG_REQUESTS'):
logging.warning(
"DISABLE_LOG_REQUESTS env var is deprecated. "
"Use ENABLE_LOG_REQUESTS instead (default: False)."
)
# Honor old behavior: if DISABLE_LOG_REQUESTS=true, don't enable logging
if os.getenv('DISABLE_LOG_REQUESTS', 'False').lower() == 'true':
args['enable_log_requests'] = False
# Add speculative decoding configuration if present
speculative_config = get_speculative_config()
if speculative_config:
args["speculative_config"] = speculative_config
return AsyncEngineArgs(**args)
+42 -9
View File
@@ -1,22 +1,55 @@
import os
import sys
import multiprocessing
import traceback
import runpod
from utils import JobInput
from engine import vLLMEngine, OpenAIvLLMEngine
from runpod import RunPodLogger
log = RunPodLogger()
vllm_engine = None
openai_engine = None
vllm_engine = vLLMEngine()
OpenAIvLLMEngine = OpenAIvLLMEngine(vllm_engine)
async def handler(job):
try:
from utils import JobInput
job_input = JobInput(job["input"])
engine = OpenAIvLLMEngine if job_input.openai_route else vllm_engine
engine = openai_engine if job_input.openai_route else vllm_engine
results_generator = engine.generate(job_input)
async for batch in results_generator:
yield batch
except Exception as e:
error_str = str(e)
full_traceback = traceback.format_exc()
runpod.serverless.start(
log.error(f"Error during inference: {error_str}")
log.error(f"Full traceback:\n{full_traceback}")
# CUDA errors = worker is broken, exit to let RunPod spin up a healthy one
if "CUDA" in error_str or "cuda" in error_str:
log.error("Terminating worker due to CUDA/GPU error")
sys.exit(1)
yield {"error": error_str}
# Only run in main process to prevent re-initialization when vLLM spawns worker subprocesses
if __name__ == "__main__" or multiprocessing.current_process().name == "MainProcess":
try:
from engine import vLLMEngine, OpenAIvLLMEngine
vllm_engine = vLLMEngine()
openai_engine = OpenAIvLLMEngine(vllm_engine)
log.info("vLLM engines initialized successfully")
except Exception as e:
log.error(f"Worker startup failed: {e}\n{traceback.format_exc()}")
sys.exit(1)
runpod.serverless.start(
{
"handler": handler,
"concurrency_modifier": lambda x: vllm_engine.max_concurrency,
"concurrency_modifier": lambda x: vllm_engine.max_concurrency if vllm_engine else 1,
"return_aggregate_stream": True,
}
)
)
+3 -4
View File
@@ -3,11 +3,10 @@ import logging
from http import HTTPStatus
from functools import wraps
from time import time
from vllm.entrypoints.openai.protocol import RequestResponseMetadata
try:
from vllm.utils import random_uuid
from vllm.entrypoints.openai.protocol import ErrorResponse
from vllm.entrypoints.openai.engine.protocol import ErrorResponse, ErrorInfo, RequestResponseMetadata
from vllm import SamplingParams
except ImportError:
logging.warning("Error importing vllm, skipping related imports. This is ONLY expected when baking model into docker image from a machine without GPUs")
@@ -88,9 +87,9 @@ class BatchSize:
self.current_batch_size = min(self.current_batch_size*self.batch_size_growth_factor, self.max_batch_size)
def create_error_response(message: str, err_type: str = "BadRequestError", status_code: HTTPStatus = HTTPStatus.BAD_REQUEST) -> ErrorResponse:
return ErrorResponse(message=message,
return ErrorResponse(error=ErrorInfo(message=message,
type=err_type,
code=status_code.value)
code=status_code.value))
def get_int_bool_env(env_var: str, default: bool) -> bool:
return int(os.getenv(env_var, int(default))) == 1