diff --git a/.runpod/hub.json b/.runpod/hub.json index 4cad044..c5d4874 100644 --- a/.runpod/hub.json +++ b/.runpod/hub.json @@ -9,7 +9,7 @@ "containerDiskInGb": 150, "gpuIds": "ADA_80_PRO,AMPERE_80", "gpuCount": 1, - "allowedCudaVersions": ["12.9", "12.8", "12.7", "12.6", "12.5", "12.4"], + "minimumCudaVersion": "12.4", "presets": [ { "name": "deepseek-ai/deepseek-r1-distill-llama-8b", @@ -181,15 +181,6 @@ "advanced": true } }, - { - "key": "QUANTIZATION_PARAM_PATH", - "input": { - "name": "Quantization Param Path", - "type": "string", - "description": "Path to the JSON file containing the KV cache scaling factors.", - "advanced": true - } - }, { "key": "MAX_MODEL_LEN", "input": { @@ -199,26 +190,6 @@ "advanced": true } }, - { - "key": "GUIDED_DECODING_BACKEND", - "input": { - "name": "Guided Decoding Backend", - "type": "string", - "description": "Which engine will be used for guided decoding by default.", - "options": [ - { - "label": "outlines", - "value": "outlines" - }, - { - "label": "lm-format-enforcer", - "value": "lm-format-enforcer" - } - ], - "default": "outlines", - "advanced": true - } - }, { "key": "DISTRIBUTED_EXECUTOR_BACKEND", "input": { @@ -238,16 +209,6 @@ "advanced": true } }, - { - "key": "WORKER_USE_RAY", - "input": { - "name": "Worker Use Ray", - "type": "boolean", - "description": "Deprecated, use --distributed-executor-backend=ray.", - "default": false, - "advanced": true - } - }, { "key": "RAY_WORKERS_USE_NSIGHT", "input": { @@ -307,26 +268,6 @@ "advanced": true } }, - { - "key": "USE_V2_BLOCK_MANAGER", - "input": { - "name": "Use V2 Block Manager", - "type": "boolean", - "description": "Use BlockSpaceMangerV2.", - "default": false, - "advanced": true - } - }, - { - "key": "NUM_LOOKAHEAD_SLOTS", - "input": { - "name": "Num Lookahead Slots", - "type": "number", - "description": "Experimental scheduling config necessary for speculative decoding.", - "default": 0, - "advanced": true - } - }, { "key": "SEED", "input": { @@ -412,53 +353,6 @@ "advanced": true } }, - { - "key": "ROPE_SCALING", - "input": { - "name": "RoPE Scaling", - "type": "string", - "description": "RoPE scaling configuration in JSON format.", - "advanced": true - } - }, - { - "key": "ROPE_THETA", - "input": { - "name": "RoPE Theta", - "type": "number", - "description": "RoPE theta. Use with rope_scaling.", - "advanced": true - } - }, - { - "key": "TOKENIZER_POOL_SIZE", - "input": { - "name": "Tokenizer Pool Size", - "type": "number", - "description": "Size of tokenizer pool to use for asynchronous tokenization.", - "default": 0, - "advanced": true - } - }, - { - "key": "TOKENIZER_POOL_TYPE", - "input": { - "name": "Tokenizer Pool Type", - "type": "string", - "description": "Type of tokenizer pool to use for asynchronous tokenization.", - "default": "ray", - "advanced": true - } - }, - { - "key": "TOKENIZER_POOL_EXTRA_CONFIG", - "input": { - "name": "Tokenizer Pool Extra Config", - "type": "string", - "description": "Extra config for tokenizer pool.", - "advanced": true - } - }, { "key": "ENABLE_LORA", "input": { @@ -489,16 +383,6 @@ "advanced": true } }, - { - "key": "LORA_EXTRA_VOCAB_SIZE", - "input": { - "name": "LoRA Extra Vocab Size", - "type": "number", - "description": "Maximum size of extra vocabulary for LoRA adapters.", - "default": 256, - "advanced": true - } - }, { "key": "LORA_DTYPE", "input": { @@ -527,15 +411,6 @@ "advanced": true } }, - { - "key": "LONG_LORA_SCALING_FACTORS", - "input": { - "name": "Long LoRA Scaling Factors", - "type": "string", - "description": "Specify multiple scaling factors for LoRA adapters.", - "advanced": true - } - }, { "key": "MAX_CPU_LORAS", "input": { @@ -615,6 +490,55 @@ "advanced": true } }, + { + "key": "SPECULATIVE_CONFIG", + "input": { + "name": "Speculative Config (JSON)", + "type": "string", + "description": "Full speculative decoding configuration as a JSON string. Overrides individual speculative env vars.", + "advanced": true + } + }, + { + "key": "SPECULATIVE_METHOD", + "input": { + "name": "Speculative Method", + "type": "string", + "description": "Speculative decoding method to use.", + "options": [ + { + "label": "None", + "value": "" + }, + { + "label": "Draft Model", + "value": "draft_model" + }, + { + "label": "N-gram", + "value": "ngram" + }, + { + "label": "EAGLE", + "value": "eagle" + }, + { + "label": "EAGLE3", + "value": "eagle3" + }, + { + "label": "Medusa", + "value": "medusa" + }, + { + "label": "MLP Speculator", + "value": "mlp_speculator" + } + ], + "default": "", + "advanced": true + } + }, { "key": "SPECULATIVE_MODEL", "input": { @@ -633,33 +557,6 @@ "advanced": true } }, - { - "key": "SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE", - "input": { - "name": "Speculative Draft Tensor Parallel Size", - "type": "number", - "description": "Number of tensor parallel replicas for the draft model.", - "advanced": true - } - }, - { - "key": "SPECULATIVE_MAX_MODEL_LEN", - "input": { - "name": "Speculative Max Model Length", - "type": "number", - "description": "The maximum sequence length supported by the draft model.", - "advanced": true - } - }, - { - "key": "SPECULATIVE_DISABLE_BY_BATCH_SIZE", - "input": { - "name": "Speculative Disable by Batch Size", - "type": "number", - "description": "Disable speculative decoding if the number of enqueue requests is larger than this value.", - "advanced": true - } - }, { "key": "NGRAM_PROMPT_LOOKUP_MAX", "input": { @@ -669,53 +566,6 @@ "advanced": true } }, - { - "key": "NGRAM_PROMPT_LOOKUP_MIN", - "input": { - "name": "Ngram Prompt Lookup Min", - "type": "number", - "description": "Min size of window for ngram prompt lookup in speculative decoding.", - "advanced": true - } - }, - { - "key": "SPEC_DECODING_ACCEPTANCE_METHOD", - "input": { - "name": "Speculative Decoding Acceptance Method", - "type": "string", - "description": "Specify the acceptance method for draft token verification in speculative decoding.", - "options": [ - { - "label": "rejection_sampler", - "value": "rejection_sampler" - }, - { - "label": "typical_acceptance_sampler", - "value": "typical_acceptance_sampler" - } - ], - "default": "rejection_sampler", - "advanced": true - } - }, - { - "key": "TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD", - "input": { - "name": "Typical Acceptance Sampler Posterior Threshold", - "type": "number", - "description": "Set the lower bound threshold for the posterior probability of a token to be accepted.", - "advanced": true - } - }, - { - "key": "TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA", - "input": { - "name": "Typical Acceptance Sampler Posterior Alpha", - "type": "number", - "description": "A scaling factor for the entropy-based threshold for token acceptance.", - "advanced": true - } - }, { "key": "MODEL_LOADER_EXTRA_CONFIG", "input": { @@ -725,54 +575,6 @@ "advanced": true } }, - { - "key": "PREEMPTION_MODE", - "input": { - "name": "Preemption Mode", - "type": "string", - "description": "If 'recompute', the engine performs preemption-aware recomputation. If 'save', the engine saves activations into the CPU memory as preemption happens.", - "advanced": true - } - }, - { - "key": "PREEMPTION_CHECK_PERIOD", - "input": { - "name": "Preemption Check Period", - "type": "number", - "description": "How frequently the engine checks if a preemption happens.", - "default": 1, - "advanced": true - } - }, - { - "key": "PREEMPTION_CPU_CAPACITY", - "input": { - "name": "Preemption CPU Capacity", - "type": "number", - "description": "The percentage of CPU memory used for the saved activations.", - "default": 2, - "advanced": true - } - }, - { - "key": "MAX_LOG_LEN", - "input": { - "name": "Max Log Length", - "type": "number", - "description": "Max number of characters or ID numbers being printed in log.", - "advanced": true - } - }, - { - "key": "DISABLE_LOGGING_REQUEST", - "input": { - "name": "Disable Logging Request", - "type": "boolean", - "description": "Disable logging requests.", - "default": false, - "advanced": true - } - }, { "key": "TOKENIZER_NAME", "input": { @@ -860,6 +662,35 @@ "advanced": true } }, + { + "key": "ATTENTION_BACKEND", + "input": { + "name": "Attention Backend", + "type": "string", + "description": "Attention backend to use (e.g., FLASH_ATTN, XFORMERS, FLASHINFER).", + "advanced": true + } + }, + { + "key": "ASYNC_SCHEDULING", + "input": { + "name": "Async Scheduling", + "type": "boolean", + "description": "Enable async scheduling for improved throughput.", + "default": false, + "advanced": true + } + }, + { + "key": "STREAM_INTERVAL", + "input": { + "name": "Stream Interval", + "type": "number", + "description": "Interval in seconds between streaming responses.", + "default": 0, + "advanced": true + } + }, { "key": "DEFAULT_BATCH_SIZE", "input": { @@ -959,12 +790,12 @@ } }, { - "key": "DISABLE_LOG_REQUESTS", + "key": "ENABLE_LOG_REQUESTS", "input": { - "name": "Disable Log Requests", + "name": "Enable Log Requests", "type": "boolean", - "description": "Enables or disables vLLM request logging", - "default": true, + "description": "Enables vLLM request logging", + "default": false, "advanced": true } }, @@ -1030,6 +861,76 @@ "default": "", "advanced": true } + }, + { + "key": "TRUST_REQUEST_CHAT_TEMPLATE", + "input": { + "name": "Trust Request Chat Template", + "type": "boolean", + "description": "Allow chat templates from incoming requests to override the server default.", + "default": false, + "advanced": true + } + }, + { + "key": "RETURN_TOKENS_AS_TOKEN_IDS", + "input": { + "name": "Return Tokens as Token IDs", + "type": "boolean", + "description": "Return token IDs instead of token strings in responses.", + "default": false, + "advanced": true + } + }, + { + "key": "EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE", + "input": { + "name": "Exclude Tools When Tool Choice None", + "type": "boolean", + "description": "When tool_choice is 'none', exclude tool definitions from the prompt.", + "default": false, + "advanced": true + } + }, + { + "key": "ENABLE_PROMPT_TOKENS_DETAILS", + "input": { + "name": "Enable Prompt Tokens Details", + "type": "boolean", + "description": "Enable detailed prompt token usage in responses.", + "default": false, + "advanced": true + } + }, + { + "key": "ENABLE_FORCE_INCLUDE_USAGE", + "input": { + "name": "Enable Force Include Usage", + "type": "boolean", + "description": "Force include usage information in all streaming responses.", + "default": false, + "advanced": true + } + }, + { + "key": "ENABLE_LOG_OUTPUTS", + "input": { + "name": "Enable Log Outputs", + "type": "boolean", + "description": "Log model outputs for debugging.", + "default": false, + "advanced": true + } + }, + { + "key": "LOG_ERROR_STACK", + "input": { + "name": "Log Error Stack", + "type": "boolean", + "description": "Log full error stack traces.", + "default": false, + "advanced": true + } } ] } diff --git a/Dockerfile b/Dockerfile index 911cbb1..0851fc8 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,18 +1,24 @@ -FROM nvidia/cuda:12.9.0-base-ubuntu22.04 +FROM nvidia/cuda:12.9.1-base-ubuntu22.04 RUN apt-get update -y \ && apt-get install -y python3-pip RUN ldconfig /usr/local/cuda-12.9/compat/ -# Install Python dependencies -COPY builder/requirements.txt /requirements.txt +ENV RAY_METRICS_EXPORT_ENABLED=0 \ + RAY_DISABLE_USAGE_STATS=1 \ + TOKENIZERS_PARALLELISM=false \ + RAYON_NUM_THREADS=4 + +# Install vLLM first to avoid PyTorch version conflicts RUN --mount=type=cache,target=/root/.cache/pip \ python3 -m pip install --upgrade pip && \ - python3 -m pip install --upgrade -r /requirements.txt + python3 -m pip install "vllm[flashinfer]==0.15.0" --extra-index-url https://download.pytorch.org/whl/cu129 -# Install vLLM -RUN python3 -m pip install vllm==0.15.0 +# Install remaining Python dependencies +COPY builder/requirements.txt /requirements.txt +RUN --mount=type=cache,target=/root/.cache/pip \ + python3 -m pip install --upgrade -r /requirements.txt # Setup for Option 2: Building the Image with the Model included ARG MODEL_NAME="" @@ -21,7 +27,7 @@ ARG BASE_PATH="/runpod-volume" ARG QUANTIZATION="" ARG MODEL_REVISION="" ARG TOKENIZER_REVISION="" -ARG VLLM_NIGHTLY="true" +ARG VLLM_NIGHTLY="false" ENV MODEL_NAME=$MODEL_NAME \ MODEL_REVISION=$MODEL_REVISION \ @@ -32,7 +38,7 @@ ENV MODEL_NAME=$MODEL_NAME \ HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \ HUGGINGFACE_HUB_CACHE="${BASE_PATH}/huggingface-cache/hub" \ HF_HOME="${BASE_PATH}/huggingface-cache/hub" \ - HF_HUB_ENABLE_HF_TRANSFER=0 + HF_HUB_ENABLE_HF_TRANSFER=0 ENV PYTHONPATH="/:/vllm-workspace" diff --git a/builder/requirements.txt b/builder/requirements.txt index f2bd849..7ec7932 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -8,7 +8,7 @@ typing-extensions>=4.8.0 pydantic pydantic-settings hf-transfer -transformers>=4.57.5 +transformers>=4.56.0,<5 bitsandbytes>=0.45.0 kernels torch>=2.10.0 diff --git a/docs/configuration.md b/docs/configuration.md index 71dfdb9..cd3deb1 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -17,19 +17,14 @@ Complete guide to all environment variables and configuration options for worker | `HF_TOKEN` | - | `str` | Hugging Face token for private and gated models. | | `DTYPE` | 'auto' | ['auto', 'half', 'float16', 'bfloat16', 'float', 'float32'] | Data type for model weights and activations. | | `KV_CACHE_DTYPE` | 'auto' | ['auto', 'fp8'] | Data type for KV cache storage. | -| `QUANTIZATION_PARAM_PATH` | None | `str` | Path to the JSON file containing the KV cache scaling factors. | | `MAX_MODEL_LEN` | None | `int` | Model context length. | -| `GUIDED_DECODING_BACKEND` | 'outlines' | ['outlines', 'lm-format-enforcer'] | Which engine will be used for guided decoding by default. | | `DISTRIBUTED_EXECUTOR_BACKEND` | None | ['ray', 'mp'] | Backend to use for distributed serving. | -| `WORKER_USE_RAY` | False | `bool` | Deprecated, use --distributed-executor-backend=ray. | | `PIPELINE_PARALLEL_SIZE` | 1 | `int` | Number of pipeline stages. | | `TENSOR_PARALLEL_SIZE` | 1 | `int` | Number of tensor parallel replicas. | | `MAX_PARALLEL_LOADING_WORKERS` | None | `int` | Load model sequentially in multiple batches. | | `RAY_WORKERS_USE_NSIGHT` | False | `bool` | If specified, use nsight to profile Ray workers. | | `ENABLE_PREFIX_CACHING` | False | `bool` | Enables automatic prefix caching. | | `DISABLE_SLIDING_WINDOW` | False | `bool` | Disables sliding window, capping to sliding window size. | -| `USE_V2_BLOCK_MANAGER` | False | `bool` | Use BlockSpaceMangerV2. | -| `NUM_LOOKAHEAD_SLOTS` | 0 | `int` | Experimental scheduling config necessary for speculative decoding. | | `SEED` | 0 | `int` | Random seed for operations. | | `NUM_GPU_BLOCKS_OVERRIDE` | None | `int` | If specified, ignore GPU profiling result and use this number of GPU blocks. | | `MAX_NUM_BATCHED_TOKENS` | None | `int` | Maximum number of batched tokens per iteration. | @@ -37,11 +32,6 @@ Complete guide to all environment variables and configuration options for worker | `MAX_LOGPROBS` | 20 | `int` | Max number of log probs to return when logprobs is specified in SamplingParams. | | `DISABLE_LOG_STATS` | False | `bool` | Disable logging statistics. | | `QUANTIZATION` | None | ['awq', 'squeezellm', 'gptq', 'bitsandbytes'] | Method used to quantize the weights. | -| `ROPE_SCALING` | None | `dict` | RoPE scaling configuration in JSON format. | -| `ROPE_THETA` | None | `float` | RoPE theta. Use with rope_scaling. | -| `TOKENIZER_POOL_SIZE` | 0 | `int` | Size of tokenizer pool to use for asynchronous tokenization. | -| `TOKENIZER_POOL_TYPE` | 'ray' | `str` | Type of tokenizer pool to use for asynchronous tokenization. | -| `TOKENIZER_POOL_EXTRA_CONFIG` | None | `dict` | Extra config for tokenizer pool. | ## LoRA (Low-Rank Adaptation) Settings @@ -50,31 +40,41 @@ Complete guide to all environment variables and configuration options for worker | `ENABLE_LORA` | False | `bool` | If True, enable handling of LoRA adapters. | | `MAX_LORAS` | 1 | `int` | Max number of LoRAs in a single batch. | | `MAX_LORA_RANK` | 16 | `int` | Max LoRA rank. | -| `LORA_EXTRA_VOCAB_SIZE` | 256 | `int` | Maximum size of extra vocabulary for LoRA adapters. | | `LORA_DTYPE` | 'auto' | ['auto', 'float16', 'bfloat16', 'float32'] | Data type for LoRA. | -| `LONG_LORA_SCALING_FACTORS` | None | `tuple` | Specify multiple scaling factors for LoRA adapters. | | `MAX_CPU_LORAS` | None | `int` | Maximum number of LoRAs to store in CPU memory. | | `FULLY_SHARDED_LORAS` | False | `bool` | Enable fully sharded LoRA layers. | | `LORA_MODULES` | `[]` | `list[dict]` | Add lora adapters from Hugging Face `[{"name": "xx", "path": "xxx/xxxx", "base_model_name": "xxx/xxxx"}]` | +> **Note:** When using LoRA with serverless deployments, the OpenAI serving engines are initialized on the first request (deferred initialization) to avoid event loop conflicts. LoRA adapter count is logged at startup. + ## Speculative Decoding Settings -| Variable | Default | Type/Choices | Description | -| ------------------------------------------------ | ------------------- | --------------------------------------------------- | ----------------------------------------------------------------------------------------- | -| `SCHEDULER_DELAY_FACTOR` | 0.0 | `float` | Apply a delay before scheduling next prompt. | -| `ENABLE_CHUNKED_PREFILL` | False | `bool` | Enable chunked prefill requests. | -| `SPECULATIVE_MODEL` | None | `str` | The name of the draft model to be used in speculative decoding. | -| `NUM_SPECULATIVE_TOKENS` | None | `int` | The number of speculative tokens to sample from the draft model. | -| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | None | `int` | Number of tensor parallel replicas for the draft model. | -| `SPECULATIVE_MAX_MODEL_LEN` | None | `int` | The maximum sequence length supported by the draft model. | -| `SPECULATIVE_DISABLE_BY_BATCH_SIZE` | None | `int` | Disable speculative decoding if the number of enqueue requests is larger than this value. | -| `NGRAM_PROMPT_LOOKUP_MAX` | None | `int` | Max size of window for ngram prompt lookup in speculative decoding. | -| `NGRAM_PROMPT_LOOKUP_MIN` | None | `int` | Min size of window for ngram prompt lookup in speculative decoding. | -| `SPEC_DECODING_ACCEPTANCE_METHOD` | 'rejection_sampler' | ['rejection_sampler', 'typical_acceptance_sampler'] | Specify the acceptance method for draft token verification in speculative decoding. | -| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD` | None | `float` | Set the lower bound threshold for the posterior probability of a token to be accepted. | -| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA` | None | `float` | A scaling factor for the entropy-based threshold for token acceptance. | +Speculative decoding can be configured in two ways: -## System Performance Settings +### Option 1: JSON Configuration + +Set `SPECULATIVE_CONFIG` to a JSON string with your full speculative decoding configuration: + +```bash +SPECULATIVE_CONFIG='{"method": "ngram", "num_speculative_tokens": 5, "prompt_lookup_max": 4}' +``` + +### Option 2: Individual Environment Variables + +| Variable | Default | Type/Choices | Description | +| ---------------------------------------- | ------- | ------------------------------------------------------------------ | ----------------------------------------------------------------------------------------- | +| `SPECULATIVE_METHOD` | None | ['draft_model', 'ngram', 'eagle', 'eagle3', 'medusa', 'mlp_speculator'] | Speculative decoding method to use. | +| `SPECULATIVE_MODEL` | None | `str` | The name of the draft model to be used in speculative decoding. | +| `NUM_SPECULATIVE_TOKENS` | None | `int` | The number of speculative tokens to sample from the draft model. | +| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | None | `int` | Number of tensor parallel replicas for the draft model. | +| `SPECULATIVE_MAX_MODEL_LEN` | None | `int` | The maximum sequence length supported by the draft model. | +| `SPECULATIVE_DISABLE_BY_BATCH_SIZE` | None | `int` | Disable speculative decoding if the number of enqueue requests is larger than this value. | +| `NGRAM_PROMPT_LOOKUP_MAX` | None | `int` | Max size of window for ngram prompt lookup in speculative decoding. | +| `NGRAM_PROMPT_LOOKUP_MIN` | None | `int` | Min size of window for ngram prompt lookup in speculative decoding. | + +If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When using individual env vars without `SPECULATIVE_METHOD`, the method is auto-detected from the model name or configuration. + +## Scheduling & Performance Settings | Variable | Default | Type/Choices | Description | | ------------------------------ | ------- | --------------- | ----------------------------------------------------------------------------------------------------------------------------------- | @@ -85,7 +85,12 @@ Complete guide to all environment variables and configuration options for worker | `ENFORCE_EAGER` | False | `bool` | Always use eager-mode PyTorch. If False(`0`), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility. | | `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. | | `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. | -| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models | +| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. | +| `SCHEDULER_DELAY_FACTOR` | 0.0 | `float` | Apply a delay before scheduling next prompt. | +| `ENABLE_CHUNKED_PREFILL` | False | `bool` | Enable chunked prefill requests. | +| `ATTENTION_BACKEND` | None | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `XFORMERS`, `FLASHINFER`). | +| `ASYNC_SCHEDULING` | False | `bool` | Enable async scheduling for improved throughput. | +| `STREAM_INTERVAL` | 0 | `float` | Interval in seconds between streaming responses. | ## Tokenizer Settings @@ -107,14 +112,21 @@ The way this works is that the first request will have a batch size of `DEFAULT_ ## OpenAI Compatibility Settings -| Variable | Default | Type/Choices | Description | -| ----------------------------------- | ----------- | ---------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `RAW_OPENAI_OUTPUT` | `1` | boolean as `int` | Enables raw OpenAI SSE format string output when streaming. **Required** to be enabled (which it is by default) for OpenAI compatibility. | -| `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | `None` | `str` | Overrides the name of the served model from model repo/path to specified name, which you will then be able to use the value for the `model` parameter when making OpenAI requests | -| `OPENAI_RESPONSE_ROLE` | `assistant` | `str` | Role of the LLM's Response in OpenAI Chat Completions. | -| `ENABLE_AUTO_TOOL_CHOICE` | `false` | `bool` | Enables automatic tool selection for supported models. Set to `true` to activate. | -| `TOOL_CALL_PARSER` | `None` | `str` | Specifies the parser for tool calls. Options: `mistral`, `hermes`, `llama3_json`, `llama4_json`, `llama4_pythonic`, `granite`, `granite-20b-fc`, `deepseek_v3`, `internlm`, `jamba`, `phi4_mini_json`, `pythonic` | -| `REASONING_PARSER` | `None` | `str` | Parser for reasoning-capable models (enables reasoning mode). Examples: `deepseek_r1`, `qwen3`, `granite`, `hunyuan_a13b`. Leave unset to disable. | +| Variable | Default | Type/Choices | Description | +| --------------------------------------- | ----------- | ---------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `RAW_OPENAI_OUTPUT` | `1` | boolean as `int` | Enables raw OpenAI SSE format string output when streaming. **Required** to be enabled (which it is by default) for OpenAI compatibility. | +| `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | `None` | `str` | Overrides the name of the served model from model repo/path to specified name, which you will then be able to use the value for the `model` parameter when making OpenAI requests | +| `OPENAI_RESPONSE_ROLE` | `assistant` | `str` | Role of the LLM's Response in OpenAI Chat Completions. | +| `ENABLE_AUTO_TOOL_CHOICE` | `false` | `bool` | Enables automatic tool selection for supported models. Set to `true` to activate. | +| `TOOL_CALL_PARSER` | `None` | `str` | Specifies the parser for tool calls. Options: `mistral`, `hermes`, `llama3_json`, `llama4_json`, `llama4_pythonic`, `granite`, `granite-20b-fc`, `deepseek_v3`, `internlm`, `jamba`, `phi4_mini_json`, `pythonic` | +| `REASONING_PARSER` | `None` | `str` | Parser for reasoning-capable models (enables reasoning mode). Examples: `deepseek_r1`, `qwen3`, `granite`, `hunyuan_a13b`. Leave unset to disable. | +| `TRUST_REQUEST_CHAT_TEMPLATE` | `false` | `bool` | Allow chat templates from incoming requests to override the server default. | +| `RETURN_TOKENS_AS_TOKEN_IDS` | `false` | `bool` | Return token IDs instead of token strings in responses. | +| `EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE` | `false` | `bool` | When tool_choice is 'none', exclude tool definitions from the prompt. | +| `ENABLE_PROMPT_TOKENS_DETAILS` | `false` | `bool` | Enable detailed prompt token usage in responses. | +| `ENABLE_FORCE_INCLUDE_USAGE` | `false` | `bool` | Force include usage information in all streaming responses. | +| `ENABLE_LOG_OUTPUTS` | `false` | `bool` | Log model outputs for debugging. | +| `LOG_ERROR_STACK` | `false` | `bool` | Log full error stack traces. | ## Serverless & Concurrency Settings @@ -122,18 +134,7 @@ The way this works is that the first request will have a batch size of `DEFAULT_ | ---------------------- | ------- | ------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | | `MAX_CONCURRENCY` | `30` | `int` | Max concurrent requests per worker. vLLM has an internal queue, so you don't have to worry about limiting by VRAM, this is for improving scaling/load balancing efficiency | | `DISABLE_LOG_STATS` | False | `bool` | Enables or disables vLLM stats logging. | -| `DISABLE_LOG_REQUESTS` | False | `bool` | Enables or disables vLLM request logging. | - -## Advanced Settings - -| Variable | Default | Type | Description | -| --------------------------- | ------- | ------- | ------------------------------------------------------------------------------------------------------------------------------------------------------ | -| `MODEL_LOADER_EXTRA_CONFIG` | None | `dict` | Extra config for model loader. | -| `PREEMPTION_MODE` | None | `str` | If 'recompute', the engine performs preemption-aware recomputation. If 'save', the engine saves activations into the CPU memory as preemption happens. | -| `PREEMPTION_CHECK_PERIOD` | 1.0 | `float` | How frequently the engine checks if a preemption happens. | -| `PREEMPTION_CPU_CAPACITY` | 2 | `float` | The percentage of CPU memory used for the saved activations. | -| `DISABLE_LOGGING_REQUEST` | False | `bool` | Disable logging requests. | -| `MAX_LOG_LEN` | None | `int` | Max number of prompt characters or prompt ID numbers being printed in log. | +| `ENABLE_LOG_REQUESTS` | False | `bool` | Enables vLLM request logging. | ## Docker Build Arguments @@ -142,13 +143,37 @@ These variables are used when building custom Docker images with models baked in | Variable | Default | Type | Description | | --------------------- | ---------------- | ----- | ------------------------------------------------- | | `BASE_PATH` | `/runpod-volume` | `str` | Storage directory for huggingface cache and model | -| `WORKER_CUDA_VERSION` | `12.1.0` | `str` | CUDA version for the worker image | +| `WORKER_CUDA_VERSION` | `12.9.1` | `str` | CUDA version for the worker image | ## Deprecated Variables -⚠️ **The following variables are deprecated and will be removed in future versions:** +> **The following variables are deprecated and will be removed in future versions:** -| Old Variable | New Variable | Note | -| ---------------------------- | ------------------------ | --------------------- | -| `MAX_CONTEXT_LEN_TO_CAPTURE` | `MAX_SEQ_LEN_TO_CAPTURE` | Use new variable name | -| `kv_cache_dtype=fp8_e5m2` | `kv_cache_dtype=fp8` | Simplified fp8 format | +| Old Variable | New Variable / Migration | Note | +| --------------------------------------------------- | ----------------------------------------------- | --------------------------------------------------------------------- | +| `MAX_CONTEXT_LEN_TO_CAPTURE` | `MAX_SEQ_LEN_TO_CAPTURE` | Use new variable name | +| `kv_cache_dtype=fp8_e5m2` | `kv_cache_dtype=fp8` | Simplified fp8 format | +| `DISABLE_LOG_REQUESTS` | `ENABLE_LOG_REQUESTS` | Logic inverted: `DISABLE_LOG_REQUESTS=true` → `ENABLE_LOG_REQUESTS=false` | +| `VLLM_ATTENTION_BACKEND` | `ATTENTION_BACKEND` | Use new variable name | +| `WORKER_USE_RAY` | `DISTRIBUTED_EXECUTOR_BACKEND=ray` | Removed in vLLM 0.15.0 | +| `USE_V2_BLOCK_MANAGER` | _(removed)_ | V2 block manager is now the default | +| `NUM_LOOKAHEAD_SLOTS` | _(removed)_ | No longer a separate config | +| `ROPE_SCALING` | _(removed)_ | Use model config directly | +| `ROPE_THETA` | _(removed)_ | Use model config directly | +| `TOKENIZER_POOL_SIZE` | _(removed)_ | Removed in vLLM 0.15.0 | +| `TOKENIZER_POOL_TYPE` | _(removed)_ | Removed in vLLM 0.15.0 | +| `TOKENIZER_POOL_EXTRA_CONFIG` | _(removed)_ | Removed in vLLM 0.15.0 | +| `QUANTIZATION_PARAM_PATH` | _(removed)_ | Removed in vLLM 0.15.0 | +| `LORA_EXTRA_VOCAB_SIZE` | _(removed)_ | Removed in vLLM 0.15.0 | +| `LONG_LORA_SCALING_FACTORS` | _(removed)_ | Removed in vLLM 0.15.0 | +| `GUIDED_DECODING_BACKEND` | _(removed)_ | Removed in vLLM 0.15.0 | +| `SPEC_DECODING_ACCEPTANCE_METHOD` | `SPECULATIVE_CONFIG` (JSON) | Use JSON config | +| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD` | `SPECULATIVE_CONFIG` (JSON) | Use JSON config | +| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA` | `SPECULATIVE_CONFIG` (JSON) | Use JSON config | +| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | `SPECULATIVE_CONFIG` (JSON) or individual env | Still supported via env var | +| `PREEMPTION_MODE` | _(removed)_ | Removed in vLLM 0.15.0 | +| `PREEMPTION_CHECK_PERIOD` | _(removed)_ | Removed in vLLM 0.15.0 | +| `PREEMPTION_CPU_CAPACITY` | _(removed)_ | Removed in vLLM 0.15.0 | +| `MAX_LOG_LEN` | _(removed)_ | Removed in vLLM 0.15.0 | +| `DISABLE_LOGGING_REQUEST` | `ENABLE_LOG_REQUESTS` | Removed in vLLM 0.15.0 | +| `MAX_SEQ_LEN_TO_CAPTURE` | _(removed)_ | Removed in vLLM 0.15.0 | diff --git a/src/engine.py b/src/engine.py index 1d9f97f..6d3c859 100644 --- a/src/engine.py +++ b/src/engine.py @@ -1,22 +1,21 @@ -import os -import logging -import json import asyncio +import json +import logging +import os +import time +from typing import AsyncGenerator, Optional from dotenv import load_dotenv -from typing import AsyncGenerator, Optional -import time from vllm import AsyncLLMEngine from vllm.entrypoints.logger import RequestLogger -from vllm.entrypoints.openai.chat_completion.serving import OpenAIServingChat from vllm.entrypoints.openai.chat_completion.protocol import ChatCompletionRequest -from vllm.entrypoints.openai.completion.serving import OpenAIServingCompletion +from vllm.entrypoints.openai.chat_completion.serving import OpenAIServingChat from vllm.entrypoints.openai.completion.protocol import CompletionRequest +from vllm.entrypoints.openai.completion.serving import OpenAIServingCompletion from vllm.entrypoints.openai.engine.protocol import ErrorResponse -from vllm.entrypoints.openai.models.serving import OpenAIServingModels from vllm.entrypoints.openai.models.protocol import BaseModelPath, LoRAModulePath - +from vllm.entrypoints.openai.models.serving import OpenAIServingModels from utils import DummyRequest, JobInput, BatchSize, create_error_response from constants import DEFAULT_MAX_CONCURRENCY, DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MIN_BATCH_SIZE @@ -28,20 +27,20 @@ class vLLMEngine: load_dotenv() # For local development self.engine_args = get_engine_args() logging.info(f"Engine args: {self.engine_args}") - + # Initialize vLLM engine first self.llm = self._initialize_llm() if engine is None else engine.llm - + # Only create custom tokenizer wrapper if not using mistral tokenizer mode # For mistral models, let vLLM handle tokenizer initialization if self.engine_args.tokenizer_mode != 'mistral': - self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model, - self.engine_args.tokenizer_revision, + self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model, + self.engine_args.tokenizer_revision, self.engine_args.trust_remote_code) else: # For mistral models, we'll get the tokenizer from vLLM later self.tokenizer = None - + self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY)) self.default_batch_size = int(os.getenv("DEFAULT_BATCH_SIZE", DEFAULT_BATCH_SIZE)) self.batch_size_growth_factor = int(os.getenv("BATCH_SIZE_GROWTH_FACTOR", DEFAULT_BATCH_SIZE_GROWTH_FACTOR)) @@ -69,7 +68,7 @@ class vLLMEngine: self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template) if self.custom_chat_template and isinstance(self.custom_chat_template, str): self.tokenizer.chat_template = self.custom_chat_template - + def apply_chat_template(self, input): if isinstance(input, list): if not self.has_chat_template: @@ -80,11 +79,11 @@ class vLLMEngine: input = [{"role": "user", "content": input}] else: raise ValueError("Input must be a string or a list of messages") - + return self.tokenizer.apply_chat_template( input, tokenize=False, add_generation_prompt=True ) - + return MinimalTokenizerWrapper(tokenizer) except Exception as e: logging.error(f"Failed to create fallback tokenizer: {e}") @@ -92,7 +91,7 @@ class vLLMEngine: def dynamic_batch_size(self, current_batch_size, batch_size_growth_factor): return min(current_batch_size*batch_size_growth_factor, self.default_batch_size) - + async def generate(self, job_input: JobInput): try: async for batch in self._generate_vllm( @@ -120,11 +119,11 @@ class vLLMEngine: batch = { "choices": [{"tokens": []} for _ in range(n_responses)], } - + max_batch_size = batch_size or self.default_batch_size batch_size_growth_factor, min_batch_size = batch_size_growth_factor or self.batch_size_growth_factor, min_batch_size or self.min_batch_size batch_size = BatchSize(max_batch_size, min_batch_size, batch_size_growth_factor) - + async for request_output in results_generator: if is_first_output: # Count input tokens only once @@ -180,7 +179,11 @@ class OpenAIvLLMEngine(vLLMEngine): self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.engine_args.model self.response_role = os.getenv("OPENAI_RESPONSE_ROLE") or "assistant" self.lora_adapters = self._load_lora_adapters() - asyncio.run(self._initialize_engines()) + self._engines_initialized = False + if self.lora_adapters: + logging.info(f"Deferring OpenAI engine initialization with {len(self.lora_adapters)} LoRA adapter(s) until first request") + else: + logging.info("Deferring OpenAI engine initialization until first request") # Handle both integer and boolean string values for RAW_OPENAI_OUTPUT raw_output_env = os.getenv("RAW_OPENAI_OUTPUT", "1") if raw_output_env.lower() in ('true', 'false'): @@ -204,7 +207,13 @@ class OpenAIvLLMEngine(vLLMEngine): continue return adapters + async def _ensure_engines_initialized(self): + if not self._engines_initialized: + await self._initialize_engines() + self._engines_initialized = True + async def _initialize_engines(self): + logging.info("Initializing OpenAI serving engines...") self.base_model_paths = [ BaseModelPath(name=self.engine_args.model, model_path=self.engine_args.model) ] @@ -231,15 +240,31 @@ class OpenAIvLLMEngine(vLLMEngine): reasoning_parser=os.getenv('REASONING_PARSER', "") or None, enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true', tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None, - enable_prompt_tokens_details=False + enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true', + trust_request_chat_template=os.getenv('TRUST_REQUEST_CHAT_TEMPLATE', 'false').lower() == 'true', + return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true', + exclude_tools_when_tool_choice_none=os.getenv('EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE', 'false').lower() == 'true', + enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true', + enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true', + log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true', ) self.completion_engine = OpenAIServingCompletion( engine_client=self.llm, models=self.serving_models, request_logger=None, + return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true', + enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true', + enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true', + log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true', ) - + + if hasattr(self.chat_engine, 'warmup'): + await self.chat_engine.warmup() + + logging.info("OpenAI serving engines initialized successfully") + async def generate(self, openai_request: JobInput): + await self._ensure_engines_initialized() if openai_request.openai_route == "/v1/models": yield await self._handle_model_request() elif openai_request.openai_route in ["/v1/chat/completions", "/v1/completions"]: @@ -247,11 +272,11 @@ class OpenAIvLLMEngine(vLLMEngine): yield response else: yield create_error_response("Invalid route").model_dump() - + async def _handle_model_request(self): models = await self.serving_models.show_available_models() return models.model_dump() - + async def _handle_chat_or_completion_request(self, openai_request: JobInput): if openai_request.openai_route == "/v1/chat/completions": request_class = ChatCompletionRequest @@ -259,7 +284,7 @@ class OpenAIvLLMEngine(vLLMEngine): elif openai_request.openai_route == "/v1/completions": request_class = CompletionRequest generator_function = self.completion_engine.create_completion - + try: request = request_class( **openai_request.openai_input @@ -267,7 +292,7 @@ class OpenAIvLLMEngine(vLLMEngine): except Exception as e: yield create_error_response(str(e)).model_dump() return - + dummy_request = DummyRequest() response_generator = await generator_function(request, raw_request=dummy_request) @@ -277,7 +302,7 @@ class OpenAIvLLMEngine(vLLMEngine): batch = [] batch_token_counter = 0 batch_size = BatchSize(self.default_batch_size, self.min_batch_size, self.batch_size_growth_factor) - + async for chunk_str in response_generator: if "data" in chunk_str: if self.raw_openai_output: @@ -299,4 +324,3 @@ class OpenAIvLLMEngine(vLLMEngine): if self.raw_openai_output: batch = "".join(batch) yield batch - \ No newline at end of file diff --git a/src/engine_args.py b/src/engine_args.py index 3c6b04d..086b76a 100644 --- a/src/engine_args.py +++ b/src/engine_args.py @@ -15,7 +15,7 @@ RENAME_ARGS_MAP = { DEFAULT_ARGS = { "disable_log_stats": os.getenv('DISABLE_LOG_STATS', 'False').lower() == 'true', - "disable_log_requests": os.getenv('DISABLE_LOG_REQUESTS', 'False').lower() == 'true', + "enable_log_requests": os.getenv('ENABLE_LOG_REQUESTS', 'False').lower() == 'true', "gpu_memory_utilization": float(os.getenv('GPU_MEMORY_UTILIZATION', 0.95)), "pipeline_parallel_size": int(os.getenv('PIPELINE_PARALLEL_SIZE', 1)), "tensor_parallel_size": int(os.getenv('TENSOR_PARALLEL_SIZE', 1)), @@ -32,7 +32,6 @@ DEFAULT_ARGS = { "quantization_param_path": os.getenv('QUANTIZATION_PARAM_PATH', None), "seed": int(os.getenv('SEED', 0)), "max_model_len": int(os.getenv('MAX_MODEL_LEN', 0)) or None, - "worker_use_ray": os.getenv('WORKER_USE_RAY', 'False').lower() == 'true', "distributed_executor_backend": os.getenv('DISTRIBUTED_EXECUTOR_BACKEND', None), "max_parallel_loading_workers": int(os.getenv('MAX_PARALLEL_LOADING_WORKERS', 0)) or None, "block_size": int(os.getenv('BLOCK_SIZE', 16)), @@ -45,17 +44,12 @@ DEFAULT_ARGS = { "max_logprobs": int(os.getenv('MAX_LOGPROBS', 20)), # Default value for OpenAI Chat Completions API "revision": os.getenv('REVISION', None), "code_revision": os.getenv('CODE_REVISION', None), - "rope_scaling": os.getenv('ROPE_SCALING', None), - "rope_theta": float(os.getenv('ROPE_THETA', 0)) or None, "tokenizer_revision": os.getenv('TOKENIZER_REVISION', None), "quantization": os.getenv('QUANTIZATION', None), "enforce_eager": os.getenv('ENFORCE_EAGER', 'False').lower() == 'true', "max_context_len_to_capture": int(os.getenv('MAX_CONTEXT_LEN_TO_CAPTURE', 0)) or None, "max_seq_len_to_capture": int(os.getenv('MAX_SEQ_LEN_TO_CAPTURE', 8192)), "disable_custom_all_reduce": os.getenv('DISABLE_CUSTOM_ALL_REDUCE', 'False').lower() == 'true', - "tokenizer_pool_size": int(os.getenv('TOKENIZER_POOL_SIZE', 0)), - "tokenizer_pool_type": os.getenv('TOKENIZER_POOL_TYPE', 'ray'), - "tokenizer_pool_extra_config": os.getenv('TOKENIZER_POOL_EXTRA_CONFIG', None), "enable_lora": os.getenv('ENABLE_LORA', 'False').lower() == 'true', "max_loras": int(os.getenv('MAX_LORAS', 1)), "max_lora_rank": int(os.getenv('MAX_LORA_RANK', 16)), @@ -63,8 +57,6 @@ DEFAULT_ARGS = { "max_prompt_adapters": int(os.getenv('MAX_PROMPT_ADAPTERS', 1)), "max_prompt_adapter_token": int(os.getenv('MAX_PROMPT_ADAPTER_TOKEN', 0)), "fully_sharded_loras": os.getenv('FULLY_SHARDED_LORAS', 'False').lower() == 'true', - "lora_extra_vocab_size": int(os.getenv('LORA_EXTRA_VOCAB_SIZE', 256)), - "long_lora_scaling_factors": tuple(map(float, os.getenv('LONG_LORA_SCALING_FACTORS', '').split(','))) if os.getenv('LONG_LORA_SCALING_FACTORS') else None, "lora_dtype": os.getenv('LORA_DTYPE', 'auto'), "max_cpu_loras": int(os.getenv('MAX_CPU_LORAS', 0)) or None, "device": os.getenv('DEVICE', 'auto'), @@ -79,6 +71,9 @@ DEFAULT_ARGS = { "enable_expert_parallel": bool(os.getenv('ENABLE_EXPERT_PARALLEL', 'False').lower() == 'true'), "qlora_adapter_name_or_path": os.getenv('QLORA_ADAPTER_NAME_OR_PATH', None), "otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None), + "attention_backend": os.getenv('ATTENTION_BACKEND', None), + "async_scheduling": os.getenv('ASYNC_SCHEDULING', 'False').lower() == 'true', + "stream_interval": float(os.getenv('STREAM_INTERVAL', 0)), } @@ -192,6 +187,7 @@ def get_speculative_config(): return config return None + limit_mm_env = os.getenv('LIMIT_MM_PER_PROMPT') if limit_mm_env is not None: DEFAULT_ARGS["limit_mm_per_prompt"] = convert_limit_mm_per_prompt(limit_mm_env) @@ -211,6 +207,7 @@ def match_vllm_args(args): renamed_args = {RENAME_ARGS_MAP.get(k, k): v for k, v in args.items()} matched_args = {k: v for k, v in renamed_args.items() if k in AsyncEngineArgs.__dataclass_fields__} return {k: v for k, v in matched_args.items() if v not in [None, "", "None"]} + def get_local_args(): """ Retrieve local arguments from a JSON file. @@ -232,28 +229,29 @@ def get_local_args(): os.environ["HF_HUB_OFFLINE"] = "1" return local_args + def get_engine_args(): # Start with default args args = DEFAULT_ARGS - + # Get env args that match keys in AsyncEngineArgs args.update(os.environ) - + # Get local args if model is baked in and overwrite env args args.update(get_local_args()) - + # if args.get("TENSORIZER_URI"): TODO: add back once tensorizer is ready # args["load_format"] = "tensorizer" # args["model_loader_extra_config"] = TensorizerConfig(tensorizer_uri=args["TENSORIZER_URI"], num_readers=None) # logging.info(f"Using tensorized model from {args['TENSORIZER_URI']}") - - + + # Rename and match to vllm args args = match_vllm_args(args) if args.get("load_format") == "bitsandbytes": args["quantization"] = args["load_format"] - + # Set tensor parallel size and max parallel loading workers if more than 1 GPU is available num_gpus = device_count() if num_gpus > 1: @@ -261,7 +259,7 @@ def get_engine_args(): args["max_parallel_loading_workers"] = None if os.getenv("MAX_PARALLEL_LOADING_WORKERS"): logging.warning("Overriding MAX_PARALLEL_LOADING_WORKERS with None because more than 1 GPU is available.") - + # Deprecated env args backwards compatibility if args.get("kv_cache_dtype") == "fp8_e5m2": args["kv_cache_dtype"] = "fp8" @@ -270,6 +268,21 @@ def get_engine_args(): args["max_seq_len_to_capture"] = int(os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE")) logging.warning("Using MAX_CONTEXT_LEN_TO_CAPTURE is deprecated. Please use MAX_SEQ_LEN_TO_CAPTURE instead.") + # VLLM_ATTENTION_BACKEND env var → attention_backend arg (deprecated shim) + vllm_attn_backend = os.getenv("VLLM_ATTENTION_BACKEND") + if vllm_attn_backend and "attention_backend" not in args: + args["attention_backend"] = vllm_attn_backend + logging.warning("VLLM_ATTENTION_BACKEND is deprecated. Please use ATTENTION_BACKEND instead.") + + # DISABLE_LOG_REQUESTS → enable_log_requests (inverted, deprecated shim) + if os.getenv("DISABLE_LOG_REQUESTS") and "enable_log_requests" not in args: + args["enable_log_requests"] = os.getenv("DISABLE_LOG_REQUESTS", "False").lower() != "true" + logging.warning("DISABLE_LOG_REQUESTS is deprecated. Please use ENABLE_LOG_REQUESTS instead.") + + # Default max_num_batched_tokens to max_model_len when not explicitly set + if args.get("max_num_batched_tokens") is None and args.get("max_model_len") is not None: + args["max_num_batched_tokens"] = args["max_model_len"] + # Add speculative decoding configuration if present speculative_config = get_speculative_config() if speculative_config: diff --git a/src/handler.py b/src/handler.py index 176ec7e..25a36d9 100644 --- a/src/handler.py +++ b/src/handler.py @@ -1,22 +1,53 @@ +import multiprocessing import os +import sys +import traceback + import runpod +from runpod import RunPodLogger + from utils import JobInput from engine import vLLMEngine, OpenAIvLLMEngine -vllm_engine = vLLMEngine() -OpenAIvLLMEngine = OpenAIvLLMEngine(vllm_engine) +log = RunPodLogger() + +# Prevent re-initialization in vLLM worker subprocesses +if multiprocessing.current_process().name == "MainProcess": + vllm_engine = vLLMEngine() + openai_engine = OpenAIvLLMEngine(vllm_engine) +else: + vllm_engine = None + openai_engine = None async def handler(job): job_input = JobInput(job["input"]) - engine = OpenAIvLLMEngine if job_input.openai_route else vllm_engine - results_generator = engine.generate(job_input) - async for batch in results_generator: - yield batch + engine = openai_engine if job_input.openai_route else vllm_engine + try: + results_generator = engine.generate(job_input) + async for batch in results_generator: + yield batch + except Exception as e: + err_str = str(e) + is_cuda_error = "CUDA" in err_str or "cuda" in err_str + is_oom = "out of memory" in err_str.lower() + + if is_cuda_error and not is_oom: + log.error(f"CUDA error (non-OOM), exiting for worker recycle: {e}") + traceback.print_exc() + sys.exit(1) + elif is_oom: + log.error(f"CUDA OOM error: {e}") + traceback.print_exc() + yield {"error": str(e)} + else: + log.error(f"Handler error: {e}") + traceback.print_exc() + yield {"error": str(e)} runpod.serverless.start( { "handler": handler, - "concurrency_modifier": lambda x: vllm_engine.max_concurrency, + "concurrency_modifier": lambda x: vllm_engine.max_concurrency if vllm_engine else 1, "return_aggregate_stream": True, } -) \ No newline at end of file +)