Merge branch 'main' into feat/update-vllm-v0.15.0
This commit is contained in:
@@ -645,16 +645,6 @@
|
|||||||
"advanced": true
|
"advanced": true
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
{
|
|
||||||
"key": "MAX_SEQ_LEN_TO_CAPTURE",
|
|
||||||
"input": {
|
|
||||||
"name": "CUDA Graph Max Content Length",
|
|
||||||
"type": "number",
|
|
||||||
"description": "Maximum context length covered by CUDA graphs. If a sequence has context length larger than this, we fall back to eager mode",
|
|
||||||
"default": 8192,
|
|
||||||
"advanced": true
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{
|
{
|
||||||
"key": "DISABLE_CUSTOM_ALL_REDUCE",
|
"key": "DISABLE_CUSTOM_ALL_REDUCE",
|
||||||
"input": {
|
"input": {
|
||||||
@@ -792,16 +782,6 @@
|
|||||||
"advanced": true
|
"advanced": true
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
{
|
|
||||||
"key": "ENABLE_LOG_REQUESTS",
|
|
||||||
"input": {
|
|
||||||
"name": "Enable Log Requests",
|
|
||||||
"type": "boolean",
|
|
||||||
"description": "Enables vLLM request logging",
|
|
||||||
"default": false,
|
|
||||||
"advanced": true
|
|
||||||
}
|
|
||||||
},
|
|
||||||
{
|
{
|
||||||
"key": "ENABLE_AUTO_TOOL_CHOICE",
|
"key": "ENABLE_AUTO_TOOL_CHOICE",
|
||||||
"input": {
|
"input": {
|
||||||
|
|||||||
+15
-12
@@ -1,21 +1,17 @@
|
|||||||
FROM nvidia/cuda:12.9.1-base-ubuntu22.04
|
FROM nvidia/cuda:12.8.0-base-ubuntu22.04
|
||||||
|
|
||||||
RUN apt-get update -y \
|
RUN apt-get update -y \
|
||||||
&& apt-get install -y python3-pip
|
&& apt-get install -y python3-pip
|
||||||
|
|
||||||
RUN ldconfig /usr/local/cuda-12.9/compat/
|
RUN ldconfig /usr/local/cuda-12.8/compat/
|
||||||
|
|
||||||
ENV RAY_METRICS_EXPORT_ENABLED=0 \
|
# Install vLLM with FlashInfer - use CUDA 12.8 PyTorch wheels (compatible with vLLM 0.15.0)
|
||||||
RAY_DISABLE_USAGE_STATS=1 \
|
RUN python3 -m pip install --upgrade pip && \
|
||||||
TOKENIZERS_PARALLELISM=false \
|
python3 -m pip install "vllm[flashinfer]==0.15.0" --extra-index-url https://download.pytorch.org/whl/cu128
|
||||||
RAYON_NUM_THREADS=4
|
|
||||||
|
|
||||||
# Install vLLM first to avoid PyTorch version conflicts
|
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
|
||||||
python3 -m pip install --upgrade pip && \
|
|
||||||
python3 -m pip install "vllm[flashinfer]==0.15.0" --extra-index-url https://download.pytorch.org/whl/cu129
|
|
||||||
|
|
||||||
# Install remaining Python dependencies
|
|
||||||
|
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
||||||
COPY builder/requirements.txt /requirements.txt
|
COPY builder/requirements.txt /requirements.txt
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
python3 -m pip install --upgrade -r /requirements.txt
|
python3 -m pip install --upgrade -r /requirements.txt
|
||||||
@@ -38,7 +34,14 @@ ENV MODEL_NAME=$MODEL_NAME \
|
|||||||
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
|
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
|
||||||
HUGGINGFACE_HUB_CACHE="${BASE_PATH}/huggingface-cache/hub" \
|
HUGGINGFACE_HUB_CACHE="${BASE_PATH}/huggingface-cache/hub" \
|
||||||
HF_HOME="${BASE_PATH}/huggingface-cache/hub" \
|
HF_HOME="${BASE_PATH}/huggingface-cache/hub" \
|
||||||
HF_HUB_ENABLE_HF_TRANSFER=0
|
HF_HUB_ENABLE_HF_TRANSFER=0 \
|
||||||
|
# Suppress Ray metrics agent warnings (not needed in containerized environments)
|
||||||
|
RAY_METRICS_EXPORT_ENABLED=0 \
|
||||||
|
RAY_DISABLE_USAGE_STATS=1 \
|
||||||
|
# Prevent rayon thread pool panic in containers where ulimit -u < nproc
|
||||||
|
# (tokenizers uses Rust's rayon which tries to spawn threads = CPU cores)
|
||||||
|
TOKENIZERS_PARALLELISM=false \
|
||||||
|
RAYON_NUM_THREADS=4
|
||||||
|
|
||||||
ENV PYTHONPATH="/:/vllm-workspace"
|
ENV PYTHONPATH="/:/vllm-workspace"
|
||||||
|
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
ray
|
ray
|
||||||
pandas
|
pandas
|
||||||
pyarrow
|
pyarrow
|
||||||
runpod>=1.8,<2.0
|
runpod
|
||||||
huggingface-hub
|
huggingface-hub
|
||||||
packaging
|
packaging
|
||||||
typing-extensions>=4.8.0
|
typing-extensions>=4.8.0
|
||||||
@@ -11,5 +11,4 @@ hf-transfer
|
|||||||
transformers>=4.56.0,<5
|
transformers>=4.56.0,<5
|
||||||
bitsandbytes>=0.45.0
|
bitsandbytes>=0.45.0
|
||||||
kernels
|
kernels
|
||||||
torch>=2.9.1
|
torch-c-dlpack-ext
|
||||||
torch-c-dlpack-ext
|
|
||||||
|
|||||||
+40
-50
@@ -25,6 +25,7 @@ Complete guide to all environment variables and configuration options for worker
|
|||||||
| `RAY_WORKERS_USE_NSIGHT` | False | `bool` | If specified, use nsight to profile Ray workers. |
|
| `RAY_WORKERS_USE_NSIGHT` | False | `bool` | If specified, use nsight to profile Ray workers. |
|
||||||
| `ENABLE_PREFIX_CACHING` | False | `bool` | Enables automatic prefix caching. |
|
| `ENABLE_PREFIX_CACHING` | False | `bool` | Enables automatic prefix caching. |
|
||||||
| `DISABLE_SLIDING_WINDOW` | False | `bool` | Disables sliding window, capping to sliding window size. |
|
| `DISABLE_SLIDING_WINDOW` | False | `bool` | Disables sliding window, capping to sliding window size. |
|
||||||
|
| `NUM_LOOKAHEAD_SLOTS` | 0 | `int` | Experimental scheduling config necessary for speculative decoding. |
|
||||||
| `SEED` | 0 | `int` | Random seed for operations. |
|
| `SEED` | 0 | `int` | Random seed for operations. |
|
||||||
| `NUM_GPU_BLOCKS_OVERRIDE` | None | `int` | If specified, ignore GPU profiling result and use this number of GPU blocks. |
|
| `NUM_GPU_BLOCKS_OVERRIDE` | None | `int` | If specified, ignore GPU profiling result and use this number of GPU blocks. |
|
||||||
| `MAX_NUM_BATCHED_TOKENS` | None | `int` | Maximum number of batched tokens per iteration. |
|
| `MAX_NUM_BATCHED_TOKENS` | None | `int` | Maximum number of batched tokens per iteration. |
|
||||||
@@ -45,7 +46,7 @@ Complete guide to all environment variables and configuration options for worker
|
|||||||
| `FULLY_SHARDED_LORAS` | False | `bool` | Enable fully sharded LoRA layers. |
|
| `FULLY_SHARDED_LORAS` | False | `bool` | Enable fully sharded LoRA layers. |
|
||||||
| `LORA_MODULES` | `[]` | `list[dict]` | Add lora adapters from Hugging Face `[{"name": "xx", "path": "xxx/xxxx", "base_model_name": "xxx/xxxx"}]` |
|
| `LORA_MODULES` | `[]` | `list[dict]` | Add lora adapters from Hugging Face `[{"name": "xx", "path": "xxx/xxxx", "base_model_name": "xxx/xxxx"}]` |
|
||||||
|
|
||||||
> **Note:** When using LoRA with serverless deployments, the OpenAI serving engines are initialized on the first request (deferred initialization) to avoid event loop conflicts. LoRA adapter count is logged at startup.
|
> **Note (Serverless)**: When LoRA adapters are configured via `LORA_MODULES`, initialization is deferred to the first request to ensure compatibility with RunPod Serverless. This means the first request will include LoRA loading time. Subsequent requests are unaffected. Check logs for "LoRA mode: X adapter(s) will load on first request" at startup.
|
||||||
|
|
||||||
## Speculative Decoding Settings
|
## Speculative Decoding Settings
|
||||||
|
|
||||||
@@ -86,11 +87,9 @@ If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When
|
|||||||
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
|
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
|
||||||
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
|
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
|
||||||
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. |
|
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. |
|
||||||
| `SCHEDULER_DELAY_FACTOR` | 0.0 | `float` | Apply a delay before scheduling next prompt. |
|
| `ATTENTION_BACKEND` | `None` | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `FLASHINFER`, `TRITON_FLASH_ATTN`). Replaces deprecated `VLLM_ATTENTION_BACKEND`. |
|
||||||
| `ENABLE_CHUNKED_PREFILL` | False | `bool` | Enable chunked prefill requests. |
|
| `ASYNC_SCHEDULING` | `None` | `bool` | Enable async scheduling (overlaps engine scheduling with GPU execution). Default: enabled in vLLM 0.14.0+. Set to `false` to disable. |
|
||||||
| `ATTENTION_BACKEND` | None | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `XFORMERS`, `FLASHINFER`). |
|
| `STREAM_INTERVAL` | `1` | `int` | Controls how often to yield streaming results. Lower = more frequent updates. |
|
||||||
| `ASYNC_SCHEDULING` | False | `bool` | Enable async scheduling for improved throughput. |
|
|
||||||
| `STREAM_INTERVAL` | 0 | `float` | Interval in seconds between streaming responses. |
|
|
||||||
|
|
||||||
## Tokenizer Settings
|
## Tokenizer Settings
|
||||||
|
|
||||||
@@ -112,21 +111,21 @@ The way this works is that the first request will have a batch size of `DEFAULT_
|
|||||||
|
|
||||||
## OpenAI Compatibility Settings
|
## OpenAI Compatibility Settings
|
||||||
|
|
||||||
| Variable | Default | Type/Choices | Description |
|
| Variable | Default | Type/Choices | Description |
|
||||||
| --------------------------------------- | ----------- | ---------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
| ----------------------------------- | ----------- | ---------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||||
| `RAW_OPENAI_OUTPUT` | `1` | boolean as `int` | Enables raw OpenAI SSE format string output when streaming. **Required** to be enabled (which it is by default) for OpenAI compatibility. |
|
| `RAW_OPENAI_OUTPUT` | `1` | boolean as `int` | Enables raw OpenAI SSE format string output when streaming. **Required** to be enabled (which it is by default) for OpenAI compatibility. |
|
||||||
| `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | `None` | `str` | Overrides the name of the served model from model repo/path to specified name, which you will then be able to use the value for the `model` parameter when making OpenAI requests |
|
| `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | `None` | `str` | Overrides the name of the served model from model repo/path to specified name, which you will then be able to use the value for the `model` parameter when making OpenAI requests |
|
||||||
| `OPENAI_RESPONSE_ROLE` | `assistant` | `str` | Role of the LLM's Response in OpenAI Chat Completions. |
|
| `OPENAI_RESPONSE_ROLE` | `assistant` | `str` | Role of the LLM's Response in OpenAI Chat Completions. |
|
||||||
| `ENABLE_AUTO_TOOL_CHOICE` | `false` | `bool` | Enables automatic tool selection for supported models. Set to `true` to activate. |
|
| `ENABLE_AUTO_TOOL_CHOICE` | `false` | `bool` | Enables automatic tool selection for supported models. Set to `true` to activate. |
|
||||||
| `TOOL_CALL_PARSER` | `None` | `str` | Specifies the parser for tool calls. Options: `mistral`, `hermes`, `llama3_json`, `llama4_json`, `llama4_pythonic`, `granite`, `granite-20b-fc`, `deepseek_v3`, `internlm`, `jamba`, `phi4_mini_json`, `pythonic` |
|
| `TOOL_CALL_PARSER` | `None` | `str` | Specifies the parser for tool calls. Options: `mistral`, `hermes`, `llama3_json`, `llama4_json`, `llama4_pythonic`, `granite`, `granite-20b-fc`, `deepseek_v3`, `internlm`, `jamba`, `phi4_mini_json`, `pythonic` |
|
||||||
| `REASONING_PARSER` | `None` | `str` | Parser for reasoning-capable models (enables reasoning mode). Examples: `deepseek_r1`, `qwen3`, `granite`, `hunyuan_a13b`. Leave unset to disable. |
|
| `REASONING_PARSER` | `None` | `str` | Parser for reasoning-capable models (enables reasoning mode). Examples: `deepseek_r1`, `qwen3`, `granite`, `hunyuan_a13b`. Leave unset to disable. |
|
||||||
| `TRUST_REQUEST_CHAT_TEMPLATE` | `false` | `bool` | Allow chat templates from incoming requests to override the server default. |
|
| `TRUST_REQUEST_CHAT_TEMPLATE` | `false` | `bool` | Allow clients to send custom chat templates in API requests. **Security consideration:** Only enable if you trust your API clients. |
|
||||||
| `RETURN_TOKENS_AS_TOKEN_IDS` | `false` | `bool` | Return token IDs instead of token strings in responses. |
|
| `RETURN_TOKENS_AS_TOKEN_IDS` | `false` | `bool` | Return token IDs instead of decoded text strings in responses. |
|
||||||
| `EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE` | `false` | `bool` | When tool_choice is 'none', exclude tool definitions from the prompt. |
|
| `EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE` | `false` | `bool` | Exclude tool definitions from the prompt when `tool_choice` is set to `none`. |
|
||||||
| `ENABLE_PROMPT_TOKENS_DETAILS` | `false` | `bool` | Enable detailed prompt token usage in responses. |
|
| `ENABLE_PROMPT_TOKENS_DETAILS` | `false` | `bool` | Include detailed prompt token information in API responses. |
|
||||||
| `ENABLE_FORCE_INCLUDE_USAGE` | `false` | `bool` | Force include usage information in all streaming responses. |
|
| `ENABLE_FORCE_INCLUDE_USAGE` | `false` | `bool` | Always include usage statistics in API responses, even when not requested. |
|
||||||
| `ENABLE_LOG_OUTPUTS` | `false` | `bool` | Log model outputs for debugging. |
|
| `ENABLE_LOG_OUTPUTS` | `false` | `bool` | Log model outputs for debugging purposes. |
|
||||||
| `LOG_ERROR_STACK` | `false` | `bool` | Log full error stack traces. |
|
| `LOG_ERROR_STACK` | `false` | `bool` | Include full stack traces in error responses for debugging. |
|
||||||
|
|
||||||
## Serverless & Concurrency Settings
|
## Serverless & Concurrency Settings
|
||||||
|
|
||||||
@@ -134,7 +133,18 @@ The way this works is that the first request will have a batch size of `DEFAULT_
|
|||||||
| ---------------------- | ------- | ------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
| ---------------------- | ------- | ------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||||
| `MAX_CONCURRENCY` | `30` | `int` | Max concurrent requests per worker. vLLM has an internal queue, so you don't have to worry about limiting by VRAM, this is for improving scaling/load balancing efficiency |
|
| `MAX_CONCURRENCY` | `30` | `int` | Max concurrent requests per worker. vLLM has an internal queue, so you don't have to worry about limiting by VRAM, this is for improving scaling/load balancing efficiency |
|
||||||
| `DISABLE_LOG_STATS` | False | `bool` | Enables or disables vLLM stats logging. |
|
| `DISABLE_LOG_STATS` | False | `bool` | Enables or disables vLLM stats logging. |
|
||||||
| `ENABLE_LOG_REQUESTS` | False | `bool` | Enables vLLM request logging. |
|
| `ENABLE_LOG_REQUESTS` | False | `bool` | Enables vLLM request logging. (Replaces deprecated `DISABLE_LOG_REQUESTS` in vLLM 0.15.0) |
|
||||||
|
|
||||||
|
## Advanced Settings
|
||||||
|
|
||||||
|
| Variable | Default | Type | Description |
|
||||||
|
| --------------------------- | ------- | ------- | ------------------------------------------------------------------------------------------------------------------------------------------------------ |
|
||||||
|
| `MODEL_LOADER_EXTRA_CONFIG` | None | `dict` | Extra config for model loader. |
|
||||||
|
| `PREEMPTION_MODE` | None | `str` | If 'recompute', the engine performs preemption-aware recomputation. If 'save', the engine saves activations into the CPU memory as preemption happens. |
|
||||||
|
| `PREEMPTION_CHECK_PERIOD` | 1.0 | `float` | How frequently the engine checks if a preemption happens. |
|
||||||
|
| `PREEMPTION_CPU_CAPACITY` | 2 | `float` | The percentage of CPU memory used for the saved activations. |
|
||||||
|
| `DISABLE_LOGGING_REQUEST` | False | `bool` | Disable logging requests. |
|
||||||
|
| `MAX_LOG_LEN` | None | `int` | Max number of prompt characters or prompt ID numbers being printed in log. |
|
||||||
|
|
||||||
## Docker Build Arguments
|
## Docker Build Arguments
|
||||||
|
|
||||||
@@ -149,31 +159,11 @@ These variables are used when building custom Docker images with models baked in
|
|||||||
|
|
||||||
> **The following variables are deprecated and will be removed in future versions:**
|
> **The following variables are deprecated and will be removed in future versions:**
|
||||||
|
|
||||||
| Old Variable | New Variable / Migration | Note |
|
| Old Variable | New Variable | Note |
|
||||||
| --------------------------------------------------- | ----------------------------------------------- | --------------------------------------------------------------------- |
|
| ---------------------------- | ------------------------ | -------------------------------------------------------------------- |
|
||||||
| `MAX_CONTEXT_LEN_TO_CAPTURE` | `MAX_SEQ_LEN_TO_CAPTURE` | Use new variable name |
|
| `MAX_CONTEXT_LEN_TO_CAPTURE` | `MAX_SEQ_LEN_TO_CAPTURE` | Use new variable name |
|
||||||
| `kv_cache_dtype=fp8_e5m2` | `kv_cache_dtype=fp8` | Simplified fp8 format |
|
| `kv_cache_dtype=fp8_e5m2` | `kv_cache_dtype=fp8` | Simplified fp8 format |
|
||||||
| `DISABLE_LOG_REQUESTS` | `ENABLE_LOG_REQUESTS` | Logic inverted: `DISABLE_LOG_REQUESTS=true` → `ENABLE_LOG_REQUESTS=false` |
|
| `USE_V2_BLOCK_MANAGER` | *(removed)* | V2 block manager is now the default in vLLM 0.13.0, setting ignored |
|
||||||
| `VLLM_ATTENTION_BACKEND` | `ATTENTION_BACKEND` | Use new variable name |
|
| `VLLM_ATTENTION_BACKEND` | `ATTENTION_BACKEND` | Use new env var name (old still works with deprecation warning) |
|
||||||
| `WORKER_USE_RAY` | `DISTRIBUTED_EXECUTOR_BACKEND=ray` | Removed in vLLM 0.15.0 |
|
| `DISABLE_LOG_REQUESTS` | `ENABLE_LOG_REQUESTS` | Inverted logic in vLLM 0.15.0 (old still works with deprecation warning) |
|
||||||
| `USE_V2_BLOCK_MANAGER` | _(removed)_ | V2 block manager is now the default |
|
|
||||||
| `NUM_LOOKAHEAD_SLOTS` | _(removed)_ | No longer a separate config |
|
|
||||||
| `ROPE_SCALING` | _(removed)_ | Use model config directly |
|
|
||||||
| `ROPE_THETA` | _(removed)_ | Use model config directly |
|
|
||||||
| `TOKENIZER_POOL_SIZE` | _(removed)_ | Removed in vLLM 0.15.0 |
|
|
||||||
| `TOKENIZER_POOL_TYPE` | _(removed)_ | Removed in vLLM 0.15.0 |
|
|
||||||
| `TOKENIZER_POOL_EXTRA_CONFIG` | _(removed)_ | Removed in vLLM 0.15.0 |
|
|
||||||
| `QUANTIZATION_PARAM_PATH` | _(removed)_ | Removed in vLLM 0.15.0 |
|
|
||||||
| `LORA_EXTRA_VOCAB_SIZE` | _(removed)_ | Removed in vLLM 0.15.0 |
|
|
||||||
| `LONG_LORA_SCALING_FACTORS` | _(removed)_ | Removed in vLLM 0.15.0 |
|
|
||||||
| `GUIDED_DECODING_BACKEND` | _(removed)_ | Removed in vLLM 0.15.0 |
|
|
||||||
| `SPEC_DECODING_ACCEPTANCE_METHOD` | `SPECULATIVE_CONFIG` (JSON) | Use JSON config |
|
|
||||||
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD` | `SPECULATIVE_CONFIG` (JSON) | Use JSON config |
|
|
||||||
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA` | `SPECULATIVE_CONFIG` (JSON) | Use JSON config |
|
|
||||||
| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | `SPECULATIVE_CONFIG` (JSON) or individual env | Still supported via env var |
|
|
||||||
| `PREEMPTION_MODE` | _(removed)_ | Removed in vLLM 0.15.0 |
|
|
||||||
| `PREEMPTION_CHECK_PERIOD` | _(removed)_ | Removed in vLLM 0.15.0 |
|
|
||||||
| `PREEMPTION_CPU_CAPACITY` | _(removed)_ | Removed in vLLM 0.15.0 |
|
|
||||||
| `MAX_LOG_LEN` | _(removed)_ | Removed in vLLM 0.15.0 |
|
|
||||||
| `DISABLE_LOGGING_REQUEST` | `ENABLE_LOG_REQUESTS` | Removed in vLLM 0.15.0 |
|
|
||||||
| `MAX_SEQ_LEN_TO_CAPTURE` | _(removed)_ | Removed in vLLM 0.15.0 |
|
|
||||||
|
|||||||
+31
-13
@@ -6,7 +6,6 @@ import time
|
|||||||
from typing import AsyncGenerator, Optional
|
from typing import AsyncGenerator, Optional
|
||||||
|
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
|
|
||||||
from vllm import AsyncLLMEngine
|
from vllm import AsyncLLMEngine
|
||||||
from vllm.entrypoints.logger import RequestLogger
|
from vllm.entrypoints.logger import RequestLogger
|
||||||
from vllm.entrypoints.openai.chat_completion.protocol import ChatCompletionRequest
|
from vllm.entrypoints.openai.chat_completion.protocol import ChatCompletionRequest
|
||||||
@@ -17,10 +16,10 @@ from vllm.entrypoints.openai.engine.protocol import ErrorResponse
|
|||||||
from vllm.entrypoints.openai.models.protocol import BaseModelPath, LoRAModulePath
|
from vllm.entrypoints.openai.models.protocol import BaseModelPath, LoRAModulePath
|
||||||
from vllm.entrypoints.openai.models.serving import OpenAIServingModels
|
from vllm.entrypoints.openai.models.serving import OpenAIServingModels
|
||||||
|
|
||||||
from utils import DummyRequest, JobInput, BatchSize, create_error_response
|
from constants import DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MAX_CONCURRENCY, DEFAULT_MIN_BATCH_SIZE
|
||||||
from constants import DEFAULT_MAX_CONCURRENCY, DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MIN_BATCH_SIZE
|
|
||||||
from tokenizer import TokenizerWrapper
|
|
||||||
from engine_args import get_engine_args
|
from engine_args import get_engine_args
|
||||||
|
from tokenizer import TokenizerWrapper
|
||||||
|
from utils import BatchSize, DummyRequest, JobInput, create_error_response
|
||||||
|
|
||||||
class vLLMEngine:
|
class vLLMEngine:
|
||||||
def __init__(self, engine = None):
|
def __init__(self, engine = None):
|
||||||
@@ -179,11 +178,21 @@ class OpenAIvLLMEngine(vLLMEngine):
|
|||||||
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.engine_args.served_model_name or self.engine_args.model
|
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.engine_args.served_model_name or self.engine_args.model
|
||||||
self.response_role = os.getenv("OPENAI_RESPONSE_ROLE") or "assistant"
|
self.response_role = os.getenv("OPENAI_RESPONSE_ROLE") or "assistant"
|
||||||
self.lora_adapters = self._load_lora_adapters()
|
self.lora_adapters = self._load_lora_adapters()
|
||||||
|
|
||||||
|
# Always defer OpenAI engine initialization to the first request.
|
||||||
|
# asyncio.run() creates a temporary event loop that gets closed, but async
|
||||||
|
# components (tokenizer pool, serving engines) bind futures to that loop.
|
||||||
|
# When Runpod's serverless handler runs in its own event loop, those futures
|
||||||
|
# are "attached to a different loop" causing RuntimeError.
|
||||||
|
# This affects all configurations, not just LoRA.
|
||||||
self._engines_initialized = False
|
self._engines_initialized = False
|
||||||
if self.lora_adapters:
|
if self.lora_adapters:
|
||||||
logging.info(f"Deferring OpenAI engine initialization with {len(self.lora_adapters)} LoRA adapter(s) until first request")
|
logging.info(f"LoRA mode: {len(self.lora_adapters)} adapter(s) will load on first request")
|
||||||
|
for adapter in self.lora_adapters:
|
||||||
|
logging.info(f" - {adapter.name}: {adapter.path}")
|
||||||
else:
|
else:
|
||||||
logging.info("Deferring OpenAI engine initialization until first request")
|
logging.info("OpenAI engines will initialize on first request")
|
||||||
|
|
||||||
# Handle both integer and boolean string values for RAW_OPENAI_OUTPUT
|
# Handle both integer and boolean string values for RAW_OPENAI_OUTPUT
|
||||||
raw_output_env = os.getenv("RAW_OPENAI_OUTPUT", "1")
|
raw_output_env = os.getenv("RAW_OPENAI_OUTPUT", "1")
|
||||||
if raw_output_env.lower() in ('true', 'false'):
|
if raw_output_env.lower() in ('true', 'false'):
|
||||||
@@ -208,12 +217,21 @@ class OpenAIvLLMEngine(vLLMEngine):
|
|||||||
return adapters
|
return adapters
|
||||||
|
|
||||||
async def _ensure_engines_initialized(self):
|
async def _ensure_engines_initialized(self):
|
||||||
|
"""Initialize engines on first request to avoid event loop mismatch.
|
||||||
|
|
||||||
|
In Runpod Serverless, the startup code runs outside the handler's event
|
||||||
|
loop. Deferring initialization to the first request ensures all async
|
||||||
|
components (tokenizer pool, serving engines, LoRA state) are created in
|
||||||
|
the correct event loop context.
|
||||||
|
"""
|
||||||
if not self._engines_initialized:
|
if not self._engines_initialized:
|
||||||
|
logging.info("Initializing OpenAI serving engines...")
|
||||||
await self._initialize_engines()
|
await self._initialize_engines()
|
||||||
self._engines_initialized = True
|
self._engines_initialized = True
|
||||||
|
logging.info("OpenAI serving engines initialized successfully")
|
||||||
|
|
||||||
async def _initialize_engines(self):
|
async def _initialize_engines(self):
|
||||||
logging.info("Initializing OpenAI serving engines...")
|
self.model_config = self.llm.model_config
|
||||||
self.base_model_paths = [
|
self.base_model_paths = [
|
||||||
BaseModelPath(name=self.served_model_name, model_path=self.engine_args.model)
|
BaseModelPath(name=self.served_model_name, model_path=self.engine_args.model)
|
||||||
]
|
]
|
||||||
@@ -237,13 +255,13 @@ class OpenAIvLLMEngine(vLLMEngine):
|
|||||||
request_logger=None,
|
request_logger=None,
|
||||||
chat_template=chat_template,
|
chat_template=chat_template,
|
||||||
chat_template_content_format="auto",
|
chat_template_content_format="auto",
|
||||||
reasoning_parser=os.getenv('REASONING_PARSER', "") or None,
|
|
||||||
enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true',
|
|
||||||
tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None,
|
|
||||||
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
|
|
||||||
trust_request_chat_template=os.getenv('TRUST_REQUEST_CHAT_TEMPLATE', 'false').lower() == 'true',
|
trust_request_chat_template=os.getenv('TRUST_REQUEST_CHAT_TEMPLATE', 'false').lower() == 'true',
|
||||||
return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true',
|
return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true',
|
||||||
|
reasoning_parser=os.getenv('REASONING_PARSER', "") or "",
|
||||||
|
enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true',
|
||||||
exclude_tools_when_tool_choice_none=os.getenv('EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE', 'false').lower() == 'true',
|
exclude_tools_when_tool_choice_none=os.getenv('EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE', 'false').lower() == 'true',
|
||||||
|
tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None,
|
||||||
|
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
|
||||||
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
||||||
enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true',
|
enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true',
|
||||||
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
|
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
|
||||||
@@ -261,10 +279,10 @@ class OpenAIvLLMEngine(vLLMEngine):
|
|||||||
if hasattr(self.chat_engine, 'warmup'):
|
if hasattr(self.chat_engine, 'warmup'):
|
||||||
await self.chat_engine.warmup()
|
await self.chat_engine.warmup()
|
||||||
|
|
||||||
logging.info("OpenAI serving engines initialized successfully")
|
|
||||||
|
|
||||||
async def generate(self, openai_request: JobInput):
|
async def generate(self, openai_request: JobInput):
|
||||||
|
# Ensure engines are ready (no-op if already initialized at startup)
|
||||||
await self._ensure_engines_initialized()
|
await self._ensure_engines_initialized()
|
||||||
|
|
||||||
if openai_request.openai_route == "/v1/models":
|
if openai_request.openai_route == "/v1/models":
|
||||||
yield await self._handle_model_request()
|
yield await self._handle_model_request()
|
||||||
elif openai_request.openai_route in ["/v1/chat/completions", "/v1/completions"]:
|
elif openai_request.openai_route in ["/v1/chat/completions", "/v1/completions"]:
|
||||||
|
|||||||
@@ -15,6 +15,7 @@ RENAME_ARGS_MAP = {
|
|||||||
|
|
||||||
DEFAULT_ARGS = {
|
DEFAULT_ARGS = {
|
||||||
"disable_log_stats": os.getenv('DISABLE_LOG_STATS', 'False').lower() == 'true',
|
"disable_log_stats": os.getenv('DISABLE_LOG_STATS', 'False').lower() == 'true',
|
||||||
|
# disable_log_requests is deprecated, use enable_log_requests instead
|
||||||
"enable_log_requests": os.getenv('ENABLE_LOG_REQUESTS', 'False').lower() == 'true',
|
"enable_log_requests": os.getenv('ENABLE_LOG_REQUESTS', 'False').lower() == 'true',
|
||||||
"gpu_memory_utilization": float(os.getenv('GPU_MEMORY_UTILIZATION', 0.95)),
|
"gpu_memory_utilization": float(os.getenv('GPU_MEMORY_UTILIZATION', 0.95)),
|
||||||
"pipeline_parallel_size": int(os.getenv('PIPELINE_PARALLEL_SIZE', 1)),
|
"pipeline_parallel_size": int(os.getenv('PIPELINE_PARALLEL_SIZE', 1)),
|
||||||
@@ -39,6 +40,7 @@ DEFAULT_ARGS = {
|
|||||||
"disable_sliding_window": os.getenv('DISABLE_SLIDING_WINDOW', 'False').lower() == 'true',
|
"disable_sliding_window": os.getenv('DISABLE_SLIDING_WINDOW', 'False').lower() == 'true',
|
||||||
"swap_space": int(os.getenv('SWAP_SPACE', 4)), # GiB
|
"swap_space": int(os.getenv('SWAP_SPACE', 4)), # GiB
|
||||||
"cpu_offload_gb": int(os.getenv('CPU_OFFLOAD_GB', 0)), # GiB
|
"cpu_offload_gb": int(os.getenv('CPU_OFFLOAD_GB', 0)), # GiB
|
||||||
|
# vLLM defaults None to 2048; keep 0 as None to let vLLM auto-calculate
|
||||||
"max_num_batched_tokens": int(os.getenv('MAX_NUM_BATCHED_TOKENS', 0)) or None,
|
"max_num_batched_tokens": int(os.getenv('MAX_NUM_BATCHED_TOKENS', 0)) or None,
|
||||||
"max_num_seqs": int(os.getenv('MAX_NUM_SEQS', 256)),
|
"max_num_seqs": int(os.getenv('MAX_NUM_SEQS', 256)),
|
||||||
"max_logprobs": int(os.getenv('MAX_LOGPROBS', 20)), # Default value for OpenAI Chat Completions API
|
"max_logprobs": int(os.getenv('MAX_LOGPROBS', 20)), # Default value for OpenAI Chat Completions API
|
||||||
|
|||||||
+38
-36
@@ -1,53 +1,55 @@
|
|||||||
import multiprocessing
|
|
||||||
import os
|
|
||||||
import sys
|
import sys
|
||||||
|
import multiprocessing
|
||||||
import traceback
|
import traceback
|
||||||
|
|
||||||
import runpod
|
import runpod
|
||||||
from runpod import RunPodLogger
|
from runpod import RunPodLogger
|
||||||
|
|
||||||
from utils import JobInput
|
|
||||||
from engine import vLLMEngine, OpenAIvLLMEngine
|
|
||||||
|
|
||||||
log = RunPodLogger()
|
log = RunPodLogger()
|
||||||
|
|
||||||
# Prevent re-initialization in vLLM worker subprocesses
|
vllm_engine = None
|
||||||
if multiprocessing.current_process().name == "MainProcess":
|
openai_engine = None
|
||||||
vllm_engine = vLLMEngine()
|
|
||||||
openai_engine = OpenAIvLLMEngine(vllm_engine)
|
|
||||||
else:
|
|
||||||
vllm_engine = None
|
|
||||||
openai_engine = None
|
|
||||||
|
|
||||||
async def handler(job):
|
async def handler(job):
|
||||||
job_input = JobInput(job["input"])
|
|
||||||
engine = openai_engine if job_input.openai_route else vllm_engine
|
|
||||||
try:
|
try:
|
||||||
|
from utils import JobInput
|
||||||
|
job_input = JobInput(job["input"])
|
||||||
|
engine = openai_engine if job_input.openai_route else vllm_engine
|
||||||
results_generator = engine.generate(job_input)
|
results_generator = engine.generate(job_input)
|
||||||
async for batch in results_generator:
|
async for batch in results_generator:
|
||||||
yield batch
|
yield batch
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
err_str = str(e)
|
error_str = str(e)
|
||||||
is_cuda_error = "CUDA" in err_str or "cuda" in err_str
|
full_traceback = traceback.format_exc()
|
||||||
is_oom = "out of memory" in err_str.lower()
|
|
||||||
|
|
||||||
if is_cuda_error and not is_oom:
|
log.error(f"Error during inference: {error_str}")
|
||||||
log.error(f"CUDA error (non-OOM), exiting for worker recycle: {e}")
|
log.error(f"Full traceback:\n{full_traceback}")
|
||||||
traceback.print_exc()
|
|
||||||
|
# CUDA errors = worker is broken, exit to let RunPod spin up a healthy one
|
||||||
|
if "CUDA" in error_str or "cuda" in error_str:
|
||||||
|
log.error("Terminating worker due to CUDA/GPU error")
|
||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
elif is_oom:
|
|
||||||
log.error(f"CUDA OOM error: {e}")
|
|
||||||
traceback.print_exc()
|
|
||||||
yield {"error": str(e)}
|
|
||||||
else:
|
|
||||||
log.error(f"Handler error: {e}")
|
|
||||||
traceback.print_exc()
|
|
||||||
yield {"error": str(e)}
|
|
||||||
|
|
||||||
runpod.serverless.start(
|
yield {"error": error_str}
|
||||||
{
|
|
||||||
"handler": handler,
|
|
||||||
"concurrency_modifier": lambda x: vllm_engine.max_concurrency if vllm_engine else 1,
|
# Only run in main process to prevent re-initialization when vLLM spawns worker subprocesses
|
||||||
"return_aggregate_stream": True,
|
if __name__ == "__main__" or multiprocessing.current_process().name == "MainProcess":
|
||||||
}
|
|
||||||
)
|
try:
|
||||||
|
from engine import vLLMEngine, OpenAIvLLMEngine
|
||||||
|
|
||||||
|
vllm_engine = vLLMEngine()
|
||||||
|
openai_engine = OpenAIvLLMEngine(vllm_engine)
|
||||||
|
log.info("vLLM engines initialized successfully")
|
||||||
|
except Exception as e:
|
||||||
|
log.error(f"Worker startup failed: {e}\n{traceback.format_exc()}")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
runpod.serverless.start(
|
||||||
|
{
|
||||||
|
"handler": handler,
|
||||||
|
"concurrency_modifier": lambda x: vllm_engine.max_concurrency if vllm_engine else 1,
|
||||||
|
"return_aggregate_stream": True,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|||||||
Reference in New Issue
Block a user