add changes from pr

This commit is contained in:
velaraptor-runpod
2026-02-06 18:10:09 -06:00
parent 461f89cea6
commit e1e9ef74ad
7 changed files with 368 additions and 368 deletions
+153 -252
View File
@@ -9,7 +9,7 @@
"containerDiskInGb": 150,
"gpuIds": "ADA_80_PRO,AMPERE_80",
"gpuCount": 1,
"allowedCudaVersions": ["12.9", "12.8", "12.7", "12.6", "12.5", "12.4"],
"minimumCudaVersion": "12.4",
"presets": [
{
"name": "deepseek-ai/deepseek-r1-distill-llama-8b",
@@ -181,15 +181,6 @@
"advanced": true
}
},
{
"key": "QUANTIZATION_PARAM_PATH",
"input": {
"name": "Quantization Param Path",
"type": "string",
"description": "Path to the JSON file containing the KV cache scaling factors.",
"advanced": true
}
},
{
"key": "MAX_MODEL_LEN",
"input": {
@@ -199,26 +190,6 @@
"advanced": true
}
},
{
"key": "GUIDED_DECODING_BACKEND",
"input": {
"name": "Guided Decoding Backend",
"type": "string",
"description": "Which engine will be used for guided decoding by default.",
"options": [
{
"label": "outlines",
"value": "outlines"
},
{
"label": "lm-format-enforcer",
"value": "lm-format-enforcer"
}
],
"default": "outlines",
"advanced": true
}
},
{
"key": "DISTRIBUTED_EXECUTOR_BACKEND",
"input": {
@@ -238,16 +209,6 @@
"advanced": true
}
},
{
"key": "WORKER_USE_RAY",
"input": {
"name": "Worker Use Ray",
"type": "boolean",
"description": "Deprecated, use --distributed-executor-backend=ray.",
"default": false,
"advanced": true
}
},
{
"key": "RAY_WORKERS_USE_NSIGHT",
"input": {
@@ -307,26 +268,6 @@
"advanced": true
}
},
{
"key": "USE_V2_BLOCK_MANAGER",
"input": {
"name": "Use V2 Block Manager",
"type": "boolean",
"description": "Use BlockSpaceMangerV2.",
"default": false,
"advanced": true
}
},
{
"key": "NUM_LOOKAHEAD_SLOTS",
"input": {
"name": "Num Lookahead Slots",
"type": "number",
"description": "Experimental scheduling config necessary for speculative decoding.",
"default": 0,
"advanced": true
}
},
{
"key": "SEED",
"input": {
@@ -412,53 +353,6 @@
"advanced": true
}
},
{
"key": "ROPE_SCALING",
"input": {
"name": "RoPE Scaling",
"type": "string",
"description": "RoPE scaling configuration in JSON format.",
"advanced": true
}
},
{
"key": "ROPE_THETA",
"input": {
"name": "RoPE Theta",
"type": "number",
"description": "RoPE theta. Use with rope_scaling.",
"advanced": true
}
},
{
"key": "TOKENIZER_POOL_SIZE",
"input": {
"name": "Tokenizer Pool Size",
"type": "number",
"description": "Size of tokenizer pool to use for asynchronous tokenization.",
"default": 0,
"advanced": true
}
},
{
"key": "TOKENIZER_POOL_TYPE",
"input": {
"name": "Tokenizer Pool Type",
"type": "string",
"description": "Type of tokenizer pool to use for asynchronous tokenization.",
"default": "ray",
"advanced": true
}
},
{
"key": "TOKENIZER_POOL_EXTRA_CONFIG",
"input": {
"name": "Tokenizer Pool Extra Config",
"type": "string",
"description": "Extra config for tokenizer pool.",
"advanced": true
}
},
{
"key": "ENABLE_LORA",
"input": {
@@ -489,16 +383,6 @@
"advanced": true
}
},
{
"key": "LORA_EXTRA_VOCAB_SIZE",
"input": {
"name": "LoRA Extra Vocab Size",
"type": "number",
"description": "Maximum size of extra vocabulary for LoRA adapters.",
"default": 256,
"advanced": true
}
},
{
"key": "LORA_DTYPE",
"input": {
@@ -527,15 +411,6 @@
"advanced": true
}
},
{
"key": "LONG_LORA_SCALING_FACTORS",
"input": {
"name": "Long LoRA Scaling Factors",
"type": "string",
"description": "Specify multiple scaling factors for LoRA adapters.",
"advanced": true
}
},
{
"key": "MAX_CPU_LORAS",
"input": {
@@ -615,6 +490,55 @@
"advanced": true
}
},
{
"key": "SPECULATIVE_CONFIG",
"input": {
"name": "Speculative Config (JSON)",
"type": "string",
"description": "Full speculative decoding configuration as a JSON string. Overrides individual speculative env vars.",
"advanced": true
}
},
{
"key": "SPECULATIVE_METHOD",
"input": {
"name": "Speculative Method",
"type": "string",
"description": "Speculative decoding method to use.",
"options": [
{
"label": "None",
"value": ""
},
{
"label": "Draft Model",
"value": "draft_model"
},
{
"label": "N-gram",
"value": "ngram"
},
{
"label": "EAGLE",
"value": "eagle"
},
{
"label": "EAGLE3",
"value": "eagle3"
},
{
"label": "Medusa",
"value": "medusa"
},
{
"label": "MLP Speculator",
"value": "mlp_speculator"
}
],
"default": "",
"advanced": true
}
},
{
"key": "SPECULATIVE_MODEL",
"input": {
@@ -633,33 +557,6 @@
"advanced": true
}
},
{
"key": "SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE",
"input": {
"name": "Speculative Draft Tensor Parallel Size",
"type": "number",
"description": "Number of tensor parallel replicas for the draft model.",
"advanced": true
}
},
{
"key": "SPECULATIVE_MAX_MODEL_LEN",
"input": {
"name": "Speculative Max Model Length",
"type": "number",
"description": "The maximum sequence length supported by the draft model.",
"advanced": true
}
},
{
"key": "SPECULATIVE_DISABLE_BY_BATCH_SIZE",
"input": {
"name": "Speculative Disable by Batch Size",
"type": "number",
"description": "Disable speculative decoding if the number of enqueue requests is larger than this value.",
"advanced": true
}
},
{
"key": "NGRAM_PROMPT_LOOKUP_MAX",
"input": {
@@ -669,53 +566,6 @@
"advanced": true
}
},
{
"key": "NGRAM_PROMPT_LOOKUP_MIN",
"input": {
"name": "Ngram Prompt Lookup Min",
"type": "number",
"description": "Min size of window for ngram prompt lookup in speculative decoding.",
"advanced": true
}
},
{
"key": "SPEC_DECODING_ACCEPTANCE_METHOD",
"input": {
"name": "Speculative Decoding Acceptance Method",
"type": "string",
"description": "Specify the acceptance method for draft token verification in speculative decoding.",
"options": [
{
"label": "rejection_sampler",
"value": "rejection_sampler"
},
{
"label": "typical_acceptance_sampler",
"value": "typical_acceptance_sampler"
}
],
"default": "rejection_sampler",
"advanced": true
}
},
{
"key": "TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD",
"input": {
"name": "Typical Acceptance Sampler Posterior Threshold",
"type": "number",
"description": "Set the lower bound threshold for the posterior probability of a token to be accepted.",
"advanced": true
}
},
{
"key": "TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA",
"input": {
"name": "Typical Acceptance Sampler Posterior Alpha",
"type": "number",
"description": "A scaling factor for the entropy-based threshold for token acceptance.",
"advanced": true
}
},
{
"key": "MODEL_LOADER_EXTRA_CONFIG",
"input": {
@@ -725,54 +575,6 @@
"advanced": true
}
},
{
"key": "PREEMPTION_MODE",
"input": {
"name": "Preemption Mode",
"type": "string",
"description": "If 'recompute', the engine performs preemption-aware recomputation. If 'save', the engine saves activations into the CPU memory as preemption happens.",
"advanced": true
}
},
{
"key": "PREEMPTION_CHECK_PERIOD",
"input": {
"name": "Preemption Check Period",
"type": "number",
"description": "How frequently the engine checks if a preemption happens.",
"default": 1,
"advanced": true
}
},
{
"key": "PREEMPTION_CPU_CAPACITY",
"input": {
"name": "Preemption CPU Capacity",
"type": "number",
"description": "The percentage of CPU memory used for the saved activations.",
"default": 2,
"advanced": true
}
},
{
"key": "MAX_LOG_LEN",
"input": {
"name": "Max Log Length",
"type": "number",
"description": "Max number of characters or ID numbers being printed in log.",
"advanced": true
}
},
{
"key": "DISABLE_LOGGING_REQUEST",
"input": {
"name": "Disable Logging Request",
"type": "boolean",
"description": "Disable logging requests.",
"default": false,
"advanced": true
}
},
{
"key": "TOKENIZER_NAME",
"input": {
@@ -860,6 +662,35 @@
"advanced": true
}
},
{
"key": "ATTENTION_BACKEND",
"input": {
"name": "Attention Backend",
"type": "string",
"description": "Attention backend to use (e.g., FLASH_ATTN, XFORMERS, FLASHINFER).",
"advanced": true
}
},
{
"key": "ASYNC_SCHEDULING",
"input": {
"name": "Async Scheduling",
"type": "boolean",
"description": "Enable async scheduling for improved throughput.",
"default": false,
"advanced": true
}
},
{
"key": "STREAM_INTERVAL",
"input": {
"name": "Stream Interval",
"type": "number",
"description": "Interval in seconds between streaming responses.",
"default": 0,
"advanced": true
}
},
{
"key": "DEFAULT_BATCH_SIZE",
"input": {
@@ -959,12 +790,12 @@
}
},
{
"key": "DISABLE_LOG_REQUESTS",
"key": "ENABLE_LOG_REQUESTS",
"input": {
"name": "Disable Log Requests",
"name": "Enable Log Requests",
"type": "boolean",
"description": "Enables or disables vLLM request logging",
"default": true,
"description": "Enables vLLM request logging",
"default": false,
"advanced": true
}
},
@@ -1030,6 +861,76 @@
"default": "",
"advanced": true
}
},
{
"key": "TRUST_REQUEST_CHAT_TEMPLATE",
"input": {
"name": "Trust Request Chat Template",
"type": "boolean",
"description": "Allow chat templates from incoming requests to override the server default.",
"default": false,
"advanced": true
}
},
{
"key": "RETURN_TOKENS_AS_TOKEN_IDS",
"input": {
"name": "Return Tokens as Token IDs",
"type": "boolean",
"description": "Return token IDs instead of token strings in responses.",
"default": false,
"advanced": true
}
},
{
"key": "EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE",
"input": {
"name": "Exclude Tools When Tool Choice None",
"type": "boolean",
"description": "When tool_choice is 'none', exclude tool definitions from the prompt.",
"default": false,
"advanced": true
}
},
{
"key": "ENABLE_PROMPT_TOKENS_DETAILS",
"input": {
"name": "Enable Prompt Tokens Details",
"type": "boolean",
"description": "Enable detailed prompt token usage in responses.",
"default": false,
"advanced": true
}
},
{
"key": "ENABLE_FORCE_INCLUDE_USAGE",
"input": {
"name": "Enable Force Include Usage",
"type": "boolean",
"description": "Force include usage information in all streaming responses.",
"default": false,
"advanced": true
}
},
{
"key": "ENABLE_LOG_OUTPUTS",
"input": {
"name": "Enable Log Outputs",
"type": "boolean",
"description": "Log model outputs for debugging.",
"default": false,
"advanced": true
}
},
{
"key": "LOG_ERROR_STACK",
"input": {
"name": "Log Error Stack",
"type": "boolean",
"description": "Log full error stack traces.",
"default": false,
"advanced": true
}
}
]
}
+14 -8
View File
@@ -1,18 +1,24 @@
FROM nvidia/cuda:12.9.0-base-ubuntu22.04
FROM nvidia/cuda:12.9.1-base-ubuntu22.04
RUN apt-get update -y \
&& apt-get install -y python3-pip
RUN ldconfig /usr/local/cuda-12.9/compat/
# Install Python dependencies
COPY builder/requirements.txt /requirements.txt
ENV RAY_METRICS_EXPORT_ENABLED=0 \
RAY_DISABLE_USAGE_STATS=1 \
TOKENIZERS_PARALLELISM=false \
RAYON_NUM_THREADS=4
# Install vLLM first to avoid PyTorch version conflicts
RUN --mount=type=cache,target=/root/.cache/pip \
python3 -m pip install --upgrade pip && \
python3 -m pip install --upgrade -r /requirements.txt
python3 -m pip install "vllm[flashinfer]==0.15.0" --extra-index-url https://download.pytorch.org/whl/cu129
# Install vLLM
RUN python3 -m pip install vllm==0.15.0
# Install remaining Python dependencies
COPY builder/requirements.txt /requirements.txt
RUN --mount=type=cache,target=/root/.cache/pip \
python3 -m pip install --upgrade -r /requirements.txt
# Setup for Option 2: Building the Image with the Model included
ARG MODEL_NAME=""
@@ -21,7 +27,7 @@ ARG BASE_PATH="/runpod-volume"
ARG QUANTIZATION=""
ARG MODEL_REVISION=""
ARG TOKENIZER_REVISION=""
ARG VLLM_NIGHTLY="true"
ARG VLLM_NIGHTLY="false"
ENV MODEL_NAME=$MODEL_NAME \
MODEL_REVISION=$MODEL_REVISION \
@@ -32,7 +38,7 @@ ENV MODEL_NAME=$MODEL_NAME \
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
HUGGINGFACE_HUB_CACHE="${BASE_PATH}/huggingface-cache/hub" \
HF_HOME="${BASE_PATH}/huggingface-cache/hub" \
HF_HUB_ENABLE_HF_TRANSFER=0
HF_HUB_ENABLE_HF_TRANSFER=0
ENV PYTHONPATH="/:/vllm-workspace"
+1 -1
View File
@@ -8,7 +8,7 @@ typing-extensions>=4.8.0
pydantic
pydantic-settings
hf-transfer
transformers>=4.57.5
transformers>=4.56.0,<5
bitsandbytes>=0.45.0
kernels
torch>=2.10.0
+79 -54
View File
@@ -17,19 +17,14 @@ Complete guide to all environment variables and configuration options for worker
| `HF_TOKEN` | - | `str` | Hugging Face token for private and gated models. |
| `DTYPE` | 'auto' | ['auto', 'half', 'float16', 'bfloat16', 'float', 'float32'] | Data type for model weights and activations. |
| `KV_CACHE_DTYPE` | 'auto' | ['auto', 'fp8'] | Data type for KV cache storage. |
| `QUANTIZATION_PARAM_PATH` | None | `str` | Path to the JSON file containing the KV cache scaling factors. |
| `MAX_MODEL_LEN` | None | `int` | Model context length. |
| `GUIDED_DECODING_BACKEND` | 'outlines' | ['outlines', 'lm-format-enforcer'] | Which engine will be used for guided decoding by default. |
| `DISTRIBUTED_EXECUTOR_BACKEND` | None | ['ray', 'mp'] | Backend to use for distributed serving. |
| `WORKER_USE_RAY` | False | `bool` | Deprecated, use --distributed-executor-backend=ray. |
| `PIPELINE_PARALLEL_SIZE` | 1 | `int` | Number of pipeline stages. |
| `TENSOR_PARALLEL_SIZE` | 1 | `int` | Number of tensor parallel replicas. |
| `MAX_PARALLEL_LOADING_WORKERS` | None | `int` | Load model sequentially in multiple batches. |
| `RAY_WORKERS_USE_NSIGHT` | False | `bool` | If specified, use nsight to profile Ray workers. |
| `ENABLE_PREFIX_CACHING` | False | `bool` | Enables automatic prefix caching. |
| `DISABLE_SLIDING_WINDOW` | False | `bool` | Disables sliding window, capping to sliding window size. |
| `USE_V2_BLOCK_MANAGER` | False | `bool` | Use BlockSpaceMangerV2. |
| `NUM_LOOKAHEAD_SLOTS` | 0 | `int` | Experimental scheduling config necessary for speculative decoding. |
| `SEED` | 0 | `int` | Random seed for operations. |
| `NUM_GPU_BLOCKS_OVERRIDE` | None | `int` | If specified, ignore GPU profiling result and use this number of GPU blocks. |
| `MAX_NUM_BATCHED_TOKENS` | None | `int` | Maximum number of batched tokens per iteration. |
@@ -37,11 +32,6 @@ Complete guide to all environment variables and configuration options for worker
| `MAX_LOGPROBS` | 20 | `int` | Max number of log probs to return when logprobs is specified in SamplingParams. |
| `DISABLE_LOG_STATS` | False | `bool` | Disable logging statistics. |
| `QUANTIZATION` | None | ['awq', 'squeezellm', 'gptq', 'bitsandbytes'] | Method used to quantize the weights. |
| `ROPE_SCALING` | None | `dict` | RoPE scaling configuration in JSON format. |
| `ROPE_THETA` | None | `float` | RoPE theta. Use with rope_scaling. |
| `TOKENIZER_POOL_SIZE` | 0 | `int` | Size of tokenizer pool to use for asynchronous tokenization. |
| `TOKENIZER_POOL_TYPE` | 'ray' | `str` | Type of tokenizer pool to use for asynchronous tokenization. |
| `TOKENIZER_POOL_EXTRA_CONFIG` | None | `dict` | Extra config for tokenizer pool. |
## LoRA (Low-Rank Adaptation) Settings
@@ -50,31 +40,41 @@ Complete guide to all environment variables and configuration options for worker
| `ENABLE_LORA` | False | `bool` | If True, enable handling of LoRA adapters. |
| `MAX_LORAS` | 1 | `int` | Max number of LoRAs in a single batch. |
| `MAX_LORA_RANK` | 16 | `int` | Max LoRA rank. |
| `LORA_EXTRA_VOCAB_SIZE` | 256 | `int` | Maximum size of extra vocabulary for LoRA adapters. |
| `LORA_DTYPE` | 'auto' | ['auto', 'float16', 'bfloat16', 'float32'] | Data type for LoRA. |
| `LONG_LORA_SCALING_FACTORS` | None | `tuple` | Specify multiple scaling factors for LoRA adapters. |
| `MAX_CPU_LORAS` | None | `int` | Maximum number of LoRAs to store in CPU memory. |
| `FULLY_SHARDED_LORAS` | False | `bool` | Enable fully sharded LoRA layers. |
| `LORA_MODULES` | `[]` | `list[dict]` | Add lora adapters from Hugging Face `[{"name": "xx", "path": "xxx/xxxx", "base_model_name": "xxx/xxxx"}]` |
> **Note:** When using LoRA with serverless deployments, the OpenAI serving engines are initialized on the first request (deferred initialization) to avoid event loop conflicts. LoRA adapter count is logged at startup.
## Speculative Decoding Settings
| Variable | Default | Type/Choices | Description |
| ------------------------------------------------ | ------------------- | --------------------------------------------------- | ----------------------------------------------------------------------------------------- |
| `SCHEDULER_DELAY_FACTOR` | 0.0 | `float` | Apply a delay before scheduling next prompt. |
| `ENABLE_CHUNKED_PREFILL` | False | `bool` | Enable chunked prefill requests. |
| `SPECULATIVE_MODEL` | None | `str` | The name of the draft model to be used in speculative decoding. |
| `NUM_SPECULATIVE_TOKENS` | None | `int` | The number of speculative tokens to sample from the draft model. |
| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | None | `int` | Number of tensor parallel replicas for the draft model. |
| `SPECULATIVE_MAX_MODEL_LEN` | None | `int` | The maximum sequence length supported by the draft model. |
| `SPECULATIVE_DISABLE_BY_BATCH_SIZE` | None | `int` | Disable speculative decoding if the number of enqueue requests is larger than this value. |
| `NGRAM_PROMPT_LOOKUP_MAX` | None | `int` | Max size of window for ngram prompt lookup in speculative decoding. |
| `NGRAM_PROMPT_LOOKUP_MIN` | None | `int` | Min size of window for ngram prompt lookup in speculative decoding. |
| `SPEC_DECODING_ACCEPTANCE_METHOD` | 'rejection_sampler' | ['rejection_sampler', 'typical_acceptance_sampler'] | Specify the acceptance method for draft token verification in speculative decoding. |
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD` | None | `float` | Set the lower bound threshold for the posterior probability of a token to be accepted. |
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA` | None | `float` | A scaling factor for the entropy-based threshold for token acceptance. |
Speculative decoding can be configured in two ways:
## System Performance Settings
### Option 1: JSON Configuration
Set `SPECULATIVE_CONFIG` to a JSON string with your full speculative decoding configuration:
```bash
SPECULATIVE_CONFIG='{"method": "ngram", "num_speculative_tokens": 5, "prompt_lookup_max": 4}'
```
### Option 2: Individual Environment Variables
| Variable | Default | Type/Choices | Description |
| ---------------------------------------- | ------- | ------------------------------------------------------------------ | ----------------------------------------------------------------------------------------- |
| `SPECULATIVE_METHOD` | None | ['draft_model', 'ngram', 'eagle', 'eagle3', 'medusa', 'mlp_speculator'] | Speculative decoding method to use. |
| `SPECULATIVE_MODEL` | None | `str` | The name of the draft model to be used in speculative decoding. |
| `NUM_SPECULATIVE_TOKENS` | None | `int` | The number of speculative tokens to sample from the draft model. |
| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | None | `int` | Number of tensor parallel replicas for the draft model. |
| `SPECULATIVE_MAX_MODEL_LEN` | None | `int` | The maximum sequence length supported by the draft model. |
| `SPECULATIVE_DISABLE_BY_BATCH_SIZE` | None | `int` | Disable speculative decoding if the number of enqueue requests is larger than this value. |
| `NGRAM_PROMPT_LOOKUP_MAX` | None | `int` | Max size of window for ngram prompt lookup in speculative decoding. |
| `NGRAM_PROMPT_LOOKUP_MIN` | None | `int` | Min size of window for ngram prompt lookup in speculative decoding. |
If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When using individual env vars without `SPECULATIVE_METHOD`, the method is auto-detected from the model name or configuration.
## Scheduling & Performance Settings
| Variable | Default | Type/Choices | Description |
| ------------------------------ | ------- | --------------- | ----------------------------------------------------------------------------------------------------------------------------------- |
@@ -85,7 +85,12 @@ Complete guide to all environment variables and configuration options for worker
| `ENFORCE_EAGER` | False | `bool` | Always use eager-mode PyTorch. If False(`0`), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility. |
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models |
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. |
| `SCHEDULER_DELAY_FACTOR` | 0.0 | `float` | Apply a delay before scheduling next prompt. |
| `ENABLE_CHUNKED_PREFILL` | False | `bool` | Enable chunked prefill requests. |
| `ATTENTION_BACKEND` | None | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `XFORMERS`, `FLASHINFER`). |
| `ASYNC_SCHEDULING` | False | `bool` | Enable async scheduling for improved throughput. |
| `STREAM_INTERVAL` | 0 | `float` | Interval in seconds between streaming responses. |
## Tokenizer Settings
@@ -107,14 +112,21 @@ The way this works is that the first request will have a batch size of `DEFAULT_
## OpenAI Compatibility Settings
| Variable | Default | Type/Choices | Description |
| ----------------------------------- | ----------- | ---------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `RAW_OPENAI_OUTPUT` | `1` | boolean as `int` | Enables raw OpenAI SSE format string output when streaming. **Required** to be enabled (which it is by default) for OpenAI compatibility. |
| `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | `None` | `str` | Overrides the name of the served model from model repo/path to specified name, which you will then be able to use the value for the `model` parameter when making OpenAI requests |
| `OPENAI_RESPONSE_ROLE` | `assistant` | `str` | Role of the LLM's Response in OpenAI Chat Completions. |
| `ENABLE_AUTO_TOOL_CHOICE` | `false` | `bool` | Enables automatic tool selection for supported models. Set to `true` to activate. |
| `TOOL_CALL_PARSER` | `None` | `str` | Specifies the parser for tool calls. Options: `mistral`, `hermes`, `llama3_json`, `llama4_json`, `llama4_pythonic`, `granite`, `granite-20b-fc`, `deepseek_v3`, `internlm`, `jamba`, `phi4_mini_json`, `pythonic` |
| `REASONING_PARSER` | `None` | `str` | Parser for reasoning-capable models (enables reasoning mode). Examples: `deepseek_r1`, `qwen3`, `granite`, `hunyuan_a13b`. Leave unset to disable. |
| Variable | Default | Type/Choices | Description |
| --------------------------------------- | ----------- | ---------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `RAW_OPENAI_OUTPUT` | `1` | boolean as `int` | Enables raw OpenAI SSE format string output when streaming. **Required** to be enabled (which it is by default) for OpenAI compatibility. |
| `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | `None` | `str` | Overrides the name of the served model from model repo/path to specified name, which you will then be able to use the value for the `model` parameter when making OpenAI requests |
| `OPENAI_RESPONSE_ROLE` | `assistant` | `str` | Role of the LLM's Response in OpenAI Chat Completions. |
| `ENABLE_AUTO_TOOL_CHOICE` | `false` | `bool` | Enables automatic tool selection for supported models. Set to `true` to activate. |
| `TOOL_CALL_PARSER` | `None` | `str` | Specifies the parser for tool calls. Options: `mistral`, `hermes`, `llama3_json`, `llama4_json`, `llama4_pythonic`, `granite`, `granite-20b-fc`, `deepseek_v3`, `internlm`, `jamba`, `phi4_mini_json`, `pythonic` |
| `REASONING_PARSER` | `None` | `str` | Parser for reasoning-capable models (enables reasoning mode). Examples: `deepseek_r1`, `qwen3`, `granite`, `hunyuan_a13b`. Leave unset to disable. |
| `TRUST_REQUEST_CHAT_TEMPLATE` | `false` | `bool` | Allow chat templates from incoming requests to override the server default. |
| `RETURN_TOKENS_AS_TOKEN_IDS` | `false` | `bool` | Return token IDs instead of token strings in responses. |
| `EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE` | `false` | `bool` | When tool_choice is 'none', exclude tool definitions from the prompt. |
| `ENABLE_PROMPT_TOKENS_DETAILS` | `false` | `bool` | Enable detailed prompt token usage in responses. |
| `ENABLE_FORCE_INCLUDE_USAGE` | `false` | `bool` | Force include usage information in all streaming responses. |
| `ENABLE_LOG_OUTPUTS` | `false` | `bool` | Log model outputs for debugging. |
| `LOG_ERROR_STACK` | `false` | `bool` | Log full error stack traces. |
## Serverless & Concurrency Settings
@@ -122,18 +134,7 @@ The way this works is that the first request will have a batch size of `DEFAULT_
| ---------------------- | ------- | ------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `MAX_CONCURRENCY` | `30` | `int` | Max concurrent requests per worker. vLLM has an internal queue, so you don't have to worry about limiting by VRAM, this is for improving scaling/load balancing efficiency |
| `DISABLE_LOG_STATS` | False | `bool` | Enables or disables vLLM stats logging. |
| `DISABLE_LOG_REQUESTS` | False | `bool` | Enables or disables vLLM request logging. |
## Advanced Settings
| Variable | Default | Type | Description |
| --------------------------- | ------- | ------- | ------------------------------------------------------------------------------------------------------------------------------------------------------ |
| `MODEL_LOADER_EXTRA_CONFIG` | None | `dict` | Extra config for model loader. |
| `PREEMPTION_MODE` | None | `str` | If 'recompute', the engine performs preemption-aware recomputation. If 'save', the engine saves activations into the CPU memory as preemption happens. |
| `PREEMPTION_CHECK_PERIOD` | 1.0 | `float` | How frequently the engine checks if a preemption happens. |
| `PREEMPTION_CPU_CAPACITY` | 2 | `float` | The percentage of CPU memory used for the saved activations. |
| `DISABLE_LOGGING_REQUEST` | False | `bool` | Disable logging requests. |
| `MAX_LOG_LEN` | None | `int` | Max number of prompt characters or prompt ID numbers being printed in log. |
| `ENABLE_LOG_REQUESTS` | False | `bool` | Enables vLLM request logging. |
## Docker Build Arguments
@@ -142,13 +143,37 @@ These variables are used when building custom Docker images with models baked in
| Variable | Default | Type | Description |
| --------------------- | ---------------- | ----- | ------------------------------------------------- |
| `BASE_PATH` | `/runpod-volume` | `str` | Storage directory for huggingface cache and model |
| `WORKER_CUDA_VERSION` | `12.1.0` | `str` | CUDA version for the worker image |
| `WORKER_CUDA_VERSION` | `12.9.1` | `str` | CUDA version for the worker image |
## Deprecated Variables
⚠️ **The following variables are deprecated and will be removed in future versions:**
> **The following variables are deprecated and will be removed in future versions:**
| Old Variable | New Variable | Note |
| ---------------------------- | ------------------------ | --------------------- |
| `MAX_CONTEXT_LEN_TO_CAPTURE` | `MAX_SEQ_LEN_TO_CAPTURE` | Use new variable name |
| `kv_cache_dtype=fp8_e5m2` | `kv_cache_dtype=fp8` | Simplified fp8 format |
| Old Variable | New Variable / Migration | Note |
| --------------------------------------------------- | ----------------------------------------------- | --------------------------------------------------------------------- |
| `MAX_CONTEXT_LEN_TO_CAPTURE` | `MAX_SEQ_LEN_TO_CAPTURE` | Use new variable name |
| `kv_cache_dtype=fp8_e5m2` | `kv_cache_dtype=fp8` | Simplified fp8 format |
| `DISABLE_LOG_REQUESTS` | `ENABLE_LOG_REQUESTS` | Logic inverted: `DISABLE_LOG_REQUESTS=true` → `ENABLE_LOG_REQUESTS=false` |
| `VLLM_ATTENTION_BACKEND` | `ATTENTION_BACKEND` | Use new variable name |
| `WORKER_USE_RAY` | `DISTRIBUTED_EXECUTOR_BACKEND=ray` | Removed in vLLM 0.15.0 |
| `USE_V2_BLOCK_MANAGER` | _(removed)_ | V2 block manager is now the default |
| `NUM_LOOKAHEAD_SLOTS` | _(removed)_ | No longer a separate config |
| `ROPE_SCALING` | _(removed)_ | Use model config directly |
| `ROPE_THETA` | _(removed)_ | Use model config directly |
| `TOKENIZER_POOL_SIZE` | _(removed)_ | Removed in vLLM 0.15.0 |
| `TOKENIZER_POOL_TYPE` | _(removed)_ | Removed in vLLM 0.15.0 |
| `TOKENIZER_POOL_EXTRA_CONFIG` | _(removed)_ | Removed in vLLM 0.15.0 |
| `QUANTIZATION_PARAM_PATH` | _(removed)_ | Removed in vLLM 0.15.0 |
| `LORA_EXTRA_VOCAB_SIZE` | _(removed)_ | Removed in vLLM 0.15.0 |
| `LONG_LORA_SCALING_FACTORS` | _(removed)_ | Removed in vLLM 0.15.0 |
| `GUIDED_DECODING_BACKEND` | _(removed)_ | Removed in vLLM 0.15.0 |
| `SPEC_DECODING_ACCEPTANCE_METHOD` | `SPECULATIVE_CONFIG` (JSON) | Use JSON config |
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD` | `SPECULATIVE_CONFIG` (JSON) | Use JSON config |
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA` | `SPECULATIVE_CONFIG` (JSON) | Use JSON config |
| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | `SPECULATIVE_CONFIG` (JSON) or individual env | Still supported via env var |
| `PREEMPTION_MODE` | _(removed)_ | Removed in vLLM 0.15.0 |
| `PREEMPTION_CHECK_PERIOD` | _(removed)_ | Removed in vLLM 0.15.0 |
| `PREEMPTION_CPU_CAPACITY` | _(removed)_ | Removed in vLLM 0.15.0 |
| `MAX_LOG_LEN` | _(removed)_ | Removed in vLLM 0.15.0 |
| `DISABLE_LOGGING_REQUEST` | `ENABLE_LOG_REQUESTS` | Removed in vLLM 0.15.0 |
| `MAX_SEQ_LEN_TO_CAPTURE` | _(removed)_ | Removed in vLLM 0.15.0 |
+53 -29
View File
@@ -1,22 +1,21 @@
import os
import logging
import json
import asyncio
import json
import logging
import os
import time
from typing import AsyncGenerator, Optional
from dotenv import load_dotenv
from typing import AsyncGenerator, Optional
import time
from vllm import AsyncLLMEngine
from vllm.entrypoints.logger import RequestLogger
from vllm.entrypoints.openai.chat_completion.serving import OpenAIServingChat
from vllm.entrypoints.openai.chat_completion.protocol import ChatCompletionRequest
from vllm.entrypoints.openai.completion.serving import OpenAIServingCompletion
from vllm.entrypoints.openai.chat_completion.serving import OpenAIServingChat
from vllm.entrypoints.openai.completion.protocol import CompletionRequest
from vllm.entrypoints.openai.completion.serving import OpenAIServingCompletion
from vllm.entrypoints.openai.engine.protocol import ErrorResponse
from vllm.entrypoints.openai.models.serving import OpenAIServingModels
from vllm.entrypoints.openai.models.protocol import BaseModelPath, LoRAModulePath
from vllm.entrypoints.openai.models.serving import OpenAIServingModels
from utils import DummyRequest, JobInput, BatchSize, create_error_response
from constants import DEFAULT_MAX_CONCURRENCY, DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MIN_BATCH_SIZE
@@ -28,20 +27,20 @@ class vLLMEngine:
load_dotenv() # For local development
self.engine_args = get_engine_args()
logging.info(f"Engine args: {self.engine_args}")
# Initialize vLLM engine first
self.llm = self._initialize_llm() if engine is None else engine.llm
# Only create custom tokenizer wrapper if not using mistral tokenizer mode
# For mistral models, let vLLM handle tokenizer initialization
if self.engine_args.tokenizer_mode != 'mistral':
self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model,
self.engine_args.tokenizer_revision,
self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model,
self.engine_args.tokenizer_revision,
self.engine_args.trust_remote_code)
else:
# For mistral models, we'll get the tokenizer from vLLM later
self.tokenizer = None
self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
self.default_batch_size = int(os.getenv("DEFAULT_BATCH_SIZE", DEFAULT_BATCH_SIZE))
self.batch_size_growth_factor = int(os.getenv("BATCH_SIZE_GROWTH_FACTOR", DEFAULT_BATCH_SIZE_GROWTH_FACTOR))
@@ -69,7 +68,7 @@ class vLLMEngine:
self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template)
if self.custom_chat_template and isinstance(self.custom_chat_template, str):
self.tokenizer.chat_template = self.custom_chat_template
def apply_chat_template(self, input):
if isinstance(input, list):
if not self.has_chat_template:
@@ -80,11 +79,11 @@ class vLLMEngine:
input = [{"role": "user", "content": input}]
else:
raise ValueError("Input must be a string or a list of messages")
return self.tokenizer.apply_chat_template(
input, tokenize=False, add_generation_prompt=True
)
return MinimalTokenizerWrapper(tokenizer)
except Exception as e:
logging.error(f"Failed to create fallback tokenizer: {e}")
@@ -92,7 +91,7 @@ class vLLMEngine:
def dynamic_batch_size(self, current_batch_size, batch_size_growth_factor):
return min(current_batch_size*batch_size_growth_factor, self.default_batch_size)
async def generate(self, job_input: JobInput):
try:
async for batch in self._generate_vllm(
@@ -120,11 +119,11 @@ class vLLMEngine:
batch = {
"choices": [{"tokens": []} for _ in range(n_responses)],
}
max_batch_size = batch_size or self.default_batch_size
batch_size_growth_factor, min_batch_size = batch_size_growth_factor or self.batch_size_growth_factor, min_batch_size or self.min_batch_size
batch_size = BatchSize(max_batch_size, min_batch_size, batch_size_growth_factor)
async for request_output in results_generator:
if is_first_output: # Count input tokens only once
@@ -180,7 +179,11 @@ class OpenAIvLLMEngine(vLLMEngine):
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.engine_args.model
self.response_role = os.getenv("OPENAI_RESPONSE_ROLE") or "assistant"
self.lora_adapters = self._load_lora_adapters()
asyncio.run(self._initialize_engines())
self._engines_initialized = False
if self.lora_adapters:
logging.info(f"Deferring OpenAI engine initialization with {len(self.lora_adapters)} LoRA adapter(s) until first request")
else:
logging.info("Deferring OpenAI engine initialization until first request")
# Handle both integer and boolean string values for RAW_OPENAI_OUTPUT
raw_output_env = os.getenv("RAW_OPENAI_OUTPUT", "1")
if raw_output_env.lower() in ('true', 'false'):
@@ -204,7 +207,13 @@ class OpenAIvLLMEngine(vLLMEngine):
continue
return adapters
async def _ensure_engines_initialized(self):
if not self._engines_initialized:
await self._initialize_engines()
self._engines_initialized = True
async def _initialize_engines(self):
logging.info("Initializing OpenAI serving engines...")
self.base_model_paths = [
BaseModelPath(name=self.engine_args.model, model_path=self.engine_args.model)
]
@@ -231,15 +240,31 @@ class OpenAIvLLMEngine(vLLMEngine):
reasoning_parser=os.getenv('REASONING_PARSER', "") or None,
enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true',
tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None,
enable_prompt_tokens_details=False
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
trust_request_chat_template=os.getenv('TRUST_REQUEST_CHAT_TEMPLATE', 'false').lower() == 'true',
return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true',
exclude_tools_when_tool_choice_none=os.getenv('EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE', 'false').lower() == 'true',
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true',
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
)
self.completion_engine = OpenAIServingCompletion(
engine_client=self.llm,
models=self.serving_models,
request_logger=None,
return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true',
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
)
if hasattr(self.chat_engine, 'warmup'):
await self.chat_engine.warmup()
logging.info("OpenAI serving engines initialized successfully")
async def generate(self, openai_request: JobInput):
await self._ensure_engines_initialized()
if openai_request.openai_route == "/v1/models":
yield await self._handle_model_request()
elif openai_request.openai_route in ["/v1/chat/completions", "/v1/completions"]:
@@ -247,11 +272,11 @@ class OpenAIvLLMEngine(vLLMEngine):
yield response
else:
yield create_error_response("Invalid route").model_dump()
async def _handle_model_request(self):
models = await self.serving_models.show_available_models()
return models.model_dump()
async def _handle_chat_or_completion_request(self, openai_request: JobInput):
if openai_request.openai_route == "/v1/chat/completions":
request_class = ChatCompletionRequest
@@ -259,7 +284,7 @@ class OpenAIvLLMEngine(vLLMEngine):
elif openai_request.openai_route == "/v1/completions":
request_class = CompletionRequest
generator_function = self.completion_engine.create_completion
try:
request = request_class(
**openai_request.openai_input
@@ -267,7 +292,7 @@ class OpenAIvLLMEngine(vLLMEngine):
except Exception as e:
yield create_error_response(str(e)).model_dump()
return
dummy_request = DummyRequest()
response_generator = await generator_function(request, raw_request=dummy_request)
@@ -277,7 +302,7 @@ class OpenAIvLLMEngine(vLLMEngine):
batch = []
batch_token_counter = 0
batch_size = BatchSize(self.default_batch_size, self.min_batch_size, self.batch_size_growth_factor)
async for chunk_str in response_generator:
if "data" in chunk_str:
if self.raw_openai_output:
@@ -299,4 +324,3 @@ class OpenAIvLLMEngine(vLLMEngine):
if self.raw_openai_output:
batch = "".join(batch)
yield batch
+29 -16
View File
@@ -15,7 +15,7 @@ RENAME_ARGS_MAP = {
DEFAULT_ARGS = {
"disable_log_stats": os.getenv('DISABLE_LOG_STATS', 'False').lower() == 'true',
"disable_log_requests": os.getenv('DISABLE_LOG_REQUESTS', 'False').lower() == 'true',
"enable_log_requests": os.getenv('ENABLE_LOG_REQUESTS', 'False').lower() == 'true',
"gpu_memory_utilization": float(os.getenv('GPU_MEMORY_UTILIZATION', 0.95)),
"pipeline_parallel_size": int(os.getenv('PIPELINE_PARALLEL_SIZE', 1)),
"tensor_parallel_size": int(os.getenv('TENSOR_PARALLEL_SIZE', 1)),
@@ -32,7 +32,6 @@ DEFAULT_ARGS = {
"quantization_param_path": os.getenv('QUANTIZATION_PARAM_PATH', None),
"seed": int(os.getenv('SEED', 0)),
"max_model_len": int(os.getenv('MAX_MODEL_LEN', 0)) or None,
"worker_use_ray": os.getenv('WORKER_USE_RAY', 'False').lower() == 'true',
"distributed_executor_backend": os.getenv('DISTRIBUTED_EXECUTOR_BACKEND', None),
"max_parallel_loading_workers": int(os.getenv('MAX_PARALLEL_LOADING_WORKERS', 0)) or None,
"block_size": int(os.getenv('BLOCK_SIZE', 16)),
@@ -45,17 +44,12 @@ DEFAULT_ARGS = {
"max_logprobs": int(os.getenv('MAX_LOGPROBS', 20)), # Default value for OpenAI Chat Completions API
"revision": os.getenv('REVISION', None),
"code_revision": os.getenv('CODE_REVISION', None),
"rope_scaling": os.getenv('ROPE_SCALING', None),
"rope_theta": float(os.getenv('ROPE_THETA', 0)) or None,
"tokenizer_revision": os.getenv('TOKENIZER_REVISION', None),
"quantization": os.getenv('QUANTIZATION', None),
"enforce_eager": os.getenv('ENFORCE_EAGER', 'False').lower() == 'true',
"max_context_len_to_capture": int(os.getenv('MAX_CONTEXT_LEN_TO_CAPTURE', 0)) or None,
"max_seq_len_to_capture": int(os.getenv('MAX_SEQ_LEN_TO_CAPTURE', 8192)),
"disable_custom_all_reduce": os.getenv('DISABLE_CUSTOM_ALL_REDUCE', 'False').lower() == 'true',
"tokenizer_pool_size": int(os.getenv('TOKENIZER_POOL_SIZE', 0)),
"tokenizer_pool_type": os.getenv('TOKENIZER_POOL_TYPE', 'ray'),
"tokenizer_pool_extra_config": os.getenv('TOKENIZER_POOL_EXTRA_CONFIG', None),
"enable_lora": os.getenv('ENABLE_LORA', 'False').lower() == 'true',
"max_loras": int(os.getenv('MAX_LORAS', 1)),
"max_lora_rank": int(os.getenv('MAX_LORA_RANK', 16)),
@@ -63,8 +57,6 @@ DEFAULT_ARGS = {
"max_prompt_adapters": int(os.getenv('MAX_PROMPT_ADAPTERS', 1)),
"max_prompt_adapter_token": int(os.getenv('MAX_PROMPT_ADAPTER_TOKEN', 0)),
"fully_sharded_loras": os.getenv('FULLY_SHARDED_LORAS', 'False').lower() == 'true',
"lora_extra_vocab_size": int(os.getenv('LORA_EXTRA_VOCAB_SIZE', 256)),
"long_lora_scaling_factors": tuple(map(float, os.getenv('LONG_LORA_SCALING_FACTORS', '').split(','))) if os.getenv('LONG_LORA_SCALING_FACTORS') else None,
"lora_dtype": os.getenv('LORA_DTYPE', 'auto'),
"max_cpu_loras": int(os.getenv('MAX_CPU_LORAS', 0)) or None,
"device": os.getenv('DEVICE', 'auto'),
@@ -79,6 +71,9 @@ DEFAULT_ARGS = {
"enable_expert_parallel": bool(os.getenv('ENABLE_EXPERT_PARALLEL', 'False').lower() == 'true'),
"qlora_adapter_name_or_path": os.getenv('QLORA_ADAPTER_NAME_OR_PATH', None),
"otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None),
"attention_backend": os.getenv('ATTENTION_BACKEND', None),
"async_scheduling": os.getenv('ASYNC_SCHEDULING', 'False').lower() == 'true',
"stream_interval": float(os.getenv('STREAM_INTERVAL', 0)),
}
@@ -192,6 +187,7 @@ def get_speculative_config():
return config
return None
limit_mm_env = os.getenv('LIMIT_MM_PER_PROMPT')
if limit_mm_env is not None:
DEFAULT_ARGS["limit_mm_per_prompt"] = convert_limit_mm_per_prompt(limit_mm_env)
@@ -211,6 +207,7 @@ def match_vllm_args(args):
renamed_args = {RENAME_ARGS_MAP.get(k, k): v for k, v in args.items()}
matched_args = {k: v for k, v in renamed_args.items() if k in AsyncEngineArgs.__dataclass_fields__}
return {k: v for k, v in matched_args.items() if v not in [None, "", "None"]}
def get_local_args():
"""
Retrieve local arguments from a JSON file.
@@ -232,28 +229,29 @@ def get_local_args():
os.environ["HF_HUB_OFFLINE"] = "1"
return local_args
def get_engine_args():
# Start with default args
args = DEFAULT_ARGS
# Get env args that match keys in AsyncEngineArgs
args.update(os.environ)
# Get local args if model is baked in and overwrite env args
args.update(get_local_args())
# if args.get("TENSORIZER_URI"): TODO: add back once tensorizer is ready
# args["load_format"] = "tensorizer"
# args["model_loader_extra_config"] = TensorizerConfig(tensorizer_uri=args["TENSORIZER_URI"], num_readers=None)
# logging.info(f"Using tensorized model from {args['TENSORIZER_URI']}")
# Rename and match to vllm args
args = match_vllm_args(args)
if args.get("load_format") == "bitsandbytes":
args["quantization"] = args["load_format"]
# Set tensor parallel size and max parallel loading workers if more than 1 GPU is available
num_gpus = device_count()
if num_gpus > 1:
@@ -261,7 +259,7 @@ def get_engine_args():
args["max_parallel_loading_workers"] = None
if os.getenv("MAX_PARALLEL_LOADING_WORKERS"):
logging.warning("Overriding MAX_PARALLEL_LOADING_WORKERS with None because more than 1 GPU is available.")
# Deprecated env args backwards compatibility
if args.get("kv_cache_dtype") == "fp8_e5m2":
args["kv_cache_dtype"] = "fp8"
@@ -270,6 +268,21 @@ def get_engine_args():
args["max_seq_len_to_capture"] = int(os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE"))
logging.warning("Using MAX_CONTEXT_LEN_TO_CAPTURE is deprecated. Please use MAX_SEQ_LEN_TO_CAPTURE instead.")
# VLLM_ATTENTION_BACKEND env var → attention_backend arg (deprecated shim)
vllm_attn_backend = os.getenv("VLLM_ATTENTION_BACKEND")
if vllm_attn_backend and "attention_backend" not in args:
args["attention_backend"] = vllm_attn_backend
logging.warning("VLLM_ATTENTION_BACKEND is deprecated. Please use ATTENTION_BACKEND instead.")
# DISABLE_LOG_REQUESTS → enable_log_requests (inverted, deprecated shim)
if os.getenv("DISABLE_LOG_REQUESTS") and "enable_log_requests" not in args:
args["enable_log_requests"] = os.getenv("DISABLE_LOG_REQUESTS", "False").lower() != "true"
logging.warning("DISABLE_LOG_REQUESTS is deprecated. Please use ENABLE_LOG_REQUESTS instead.")
# Default max_num_batched_tokens to max_model_len when not explicitly set
if args.get("max_num_batched_tokens") is None and args.get("max_model_len") is not None:
args["max_num_batched_tokens"] = args["max_model_len"]
# Add speculative decoding configuration if present
speculative_config = get_speculative_config()
if speculative_config:
+39 -8
View File
@@ -1,22 +1,53 @@
import multiprocessing
import os
import sys
import traceback
import runpod
from runpod import RunPodLogger
from utils import JobInput
from engine import vLLMEngine, OpenAIvLLMEngine
vllm_engine = vLLMEngine()
OpenAIvLLMEngine = OpenAIvLLMEngine(vllm_engine)
log = RunPodLogger()
# Prevent re-initialization in vLLM worker subprocesses
if multiprocessing.current_process().name == "MainProcess":
vllm_engine = vLLMEngine()
openai_engine = OpenAIvLLMEngine(vllm_engine)
else:
vllm_engine = None
openai_engine = None
async def handler(job):
job_input = JobInput(job["input"])
engine = OpenAIvLLMEngine if job_input.openai_route else vllm_engine
results_generator = engine.generate(job_input)
async for batch in results_generator:
yield batch
engine = openai_engine if job_input.openai_route else vllm_engine
try:
results_generator = engine.generate(job_input)
async for batch in results_generator:
yield batch
except Exception as e:
err_str = str(e)
is_cuda_error = "CUDA" in err_str or "cuda" in err_str
is_oom = "out of memory" in err_str.lower()
if is_cuda_error and not is_oom:
log.error(f"CUDA error (non-OOM), exiting for worker recycle: {e}")
traceback.print_exc()
sys.exit(1)
elif is_oom:
log.error(f"CUDA OOM error: {e}")
traceback.print_exc()
yield {"error": str(e)}
else:
log.error(f"Handler error: {e}")
traceback.print_exc()
yield {"error": str(e)}
runpod.serverless.start(
{
"handler": handler,
"concurrency_modifier": lambda x: vllm_engine.max_concurrency,
"concurrency_modifier": lambda x: vllm_engine.max_concurrency if vllm_engine else 1,
"return_aggregate_stream": True,
}
)
)