Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9e1c483136 | ||
|
|
d71ea9939d | ||
|
|
7e2b4e2288 | ||
|
|
015f8f3c4d | ||
|
|
75ffcf73f2 | ||
|
|
d7ba3b6ab7 | ||
|
|
b11c91722c | ||
|
|
fcdc799e0d | ||
|
|
84ec446493 | ||
|
|
db246653a2 | ||
|
|
1b3228a2dc | ||
|
|
4817d4a8e7 | ||
|
|
0378382a92 | ||
|
|
08580e7ccf | ||
|
|
8868aae6b1 | ||
|
|
fb8adc5c06 | ||
|
|
7351da512b | ||
|
|
1e78043b2c | ||
|
|
352c64f4c1 | ||
|
|
0922f5b435 | ||
|
|
5d9a48fc70 | ||
|
|
5d1579e361 | ||
|
|
0488b77d89 | ||
|
|
d3a962c33b | ||
|
|
105c125698 | ||
|
|
0a0ccfcb60 | ||
|
|
c8ce53c72c | ||
|
|
9618e799ba | ||
|
|
cb3f077dba | ||
|
|
69646b9e99 | ||
|
|
8b991a7ad7 | ||
|
|
dac05b62b3 | ||
|
|
d356c31675 | ||
|
|
14b74a4989 | ||
|
|
80072047ab | ||
|
|
50aba8fb57 | ||
|
|
9edc5715ce | ||
|
|
4c91f2c5b5 | ||
|
|
6265b99348 | ||
|
|
026f8d700b | ||
|
|
146bdb0252 | ||
|
|
da01193a3d | ||
|
|
c2e6cc9f61 | ||
|
|
69968a6b39 | ||
|
|
32b29d4c6c | ||
|
|
dcea4fc4f9 | ||
|
|
9c139e8ceb | ||
|
|
678bb4be8f | ||
|
|
ed315a175e | ||
|
|
747cdf5891 | ||
|
|
22356ee2b3 |
@@ -0,0 +1,71 @@
|
||||
name: CI | Sync vLLM version in READMEs
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: ["main"]
|
||||
paths:
|
||||
- "Dockerfile"
|
||||
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
|
||||
jobs:
|
||||
sync_version:
|
||||
runs-on: ubuntu-latest
|
||||
name: Check README version matches Dockerfile and update if needed
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Extract vLLM version from Dockerfile and sync READMEs
|
||||
run: |
|
||||
echo "Extracting vLLM version from Dockerfile..."
|
||||
dockerfile_version=$(grep -oP 'vllm(?:\[[\w,]+\])?==\K[\d.]+' Dockerfile | head -1)
|
||||
|
||||
if [ -z "$dockerfile_version" ]; then
|
||||
echo "ERROR: Could not extract vLLM version from Dockerfile."
|
||||
exit 1
|
||||
fi
|
||||
echo "Dockerfile vLLM version: $dockerfile_version"
|
||||
echo "VLLM_VERSION=$dockerfile_version" >> $GITHUB_ENV
|
||||
|
||||
updated=0
|
||||
for readme in README.md .runpod/README.md; do
|
||||
if [ ! -f "$readme" ]; then
|
||||
echo "Skipping $readme (not found)"
|
||||
continue
|
||||
fi
|
||||
|
||||
readme_version=$(grep -oP 'Current vLLM version: \[\K[\d.]+' "$readme" || echo "")
|
||||
echo "$readme current version: ${readme_version:-not found}"
|
||||
|
||||
if [ "$readme_version" = "$dockerfile_version" ]; then
|
||||
echo "$readme is already up to date."
|
||||
continue
|
||||
fi
|
||||
|
||||
echo "Updating $readme from $readme_version to $dockerfile_version..."
|
||||
sed -i "s|Current vLLM version: \[${readme_version}\](https://github.com/vllm-project/vllm/releases/tag/v${readme_version})|Current vLLM version: [${dockerfile_version}](https://github.com/vllm-project/vllm/releases/tag/v${dockerfile_version})|g" "$readme"
|
||||
updated=1
|
||||
done
|
||||
|
||||
echo "UPDATED=$updated" >> $GITHUB_ENV
|
||||
|
||||
- name: Create Pull Request
|
||||
if: env.UPDATED == '1'
|
||||
uses: peter-evans/create-pull-request@v7
|
||||
with:
|
||||
token: ${{ secrets.GITHUB_TOKEN }}
|
||||
commit-message: "docs: sync vLLM version to ${{ env.VLLM_VERSION }} in READMEs"
|
||||
title: "docs: sync vLLM version to ${{ env.VLLM_VERSION }} in READMEs"
|
||||
body: |
|
||||
The vLLM version in the Dockerfile has been updated to `${{ env.VLLM_VERSION }}`.
|
||||
|
||||
This PR syncs the version badge/link in:
|
||||
- `README.md`
|
||||
- `.runpod/README.md`
|
||||
branch: docs/sync-vllm-version-${{ env.VLLM_VERSION }}
|
||||
labels: documentation
|
||||
@@ -0,0 +1,32 @@
|
||||
name: Tests
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
branches:
|
||||
- "**"
|
||||
push:
|
||||
branches:
|
||||
- "main"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
pytest:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.11"
|
||||
|
||||
- name: Install test dependencies
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
pip install -r tests/requirements.txt
|
||||
|
||||
- name: Run unit tests
|
||||
run: python -m pytest tests -v
|
||||
+12
-1
@@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
|
||||
|
||||
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
||||
|
||||
Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.19.1)
|
||||
Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
|
||||
|
||||
---
|
||||
|
||||
@@ -33,6 +33,17 @@ All behaviour is controlled through environment variables:
|
||||
|
||||
**Pass any vLLM engine arg** not listed above by setting an env var with the **UPPERCASED** field name (e.g. `MAX_MODEL_LEN=4096`, `ENABLE_CHUNKED_PREFILL=true`). The worker auto-discovers all `AsyncEngineArgs` fields from env. See the [vLLM engine args docs](https://docs.vllm.ai/en/latest/configuration/engine_args) for all available options.
|
||||
|
||||
**Configuration file:** You can also supply a `config.yaml` instead of (or alongside) env vars. Mount it at `/vllm_config.yaml` in the container, or set `VLLM_CONFIG_FILE` to a custom path. Use the same key names as `vllm serve` — hyphens and underscores both work:
|
||||
|
||||
```yaml
|
||||
model: meta-llama/Llama-3.1-8B-Instruct
|
||||
max-model-len: 8192
|
||||
gpu-memory-utilization: 0.90
|
||||
quantization: awq
|
||||
```
|
||||
|
||||
Environment variables always override config file values.
|
||||
|
||||
For complete configuration options, see the [full configuration documentation](https://github.com/runpod-workers/worker-vllm/blob/main/docs/configuration.md).
|
||||
|
||||
### Specify Transformers Version
|
||||
|
||||
+11
-1
@@ -9,7 +9,7 @@
|
||||
"containerDiskInGb": 150,
|
||||
"gpuIds": "ADA_80_PRO,AMPERE_80",
|
||||
"gpuCount": 1,
|
||||
"allowedCudaVersions": ["12.9", "12.8"],
|
||||
"allowedCudaVersions": ["13.0"],
|
||||
"presets": [
|
||||
{
|
||||
"name": "deepseek-ai/deepseek-r1-distill-llama-8b",
|
||||
@@ -805,6 +805,16 @@
|
||||
"default": "expandable_segments:True",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "VLLM_USE_DEEP_GEMM",
|
||||
"input": {
|
||||
"name": "Use DeepGEMM",
|
||||
"type": "string",
|
||||
"description": "Enable DeepGEMM FP8 kernels (MoE and MQA logits). Set to 1 to enable, 0 to disable. Required for DeepSeek V4 models. Disabled by default — enable on H100/H200 for potential throughput gains. Some GPUs (e.g. H20) may perform better with this off.",
|
||||
"default": "0",
|
||||
"advanced": true
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
"input": {
|
||||
"prompt": "Write a short poem about artificial intelligence."
|
||||
},
|
||||
"timeout": 30000
|
||||
"timeout": 300000
|
||||
},
|
||||
{
|
||||
"name": "openai_messages_test",
|
||||
@@ -26,11 +26,11 @@
|
||||
"temperature": 0.1
|
||||
}
|
||||
},
|
||||
"timeout": 30000
|
||||
"timeout": 300000
|
||||
}
|
||||
],
|
||||
"config": {
|
||||
"gpuTypeId": "NVIDIA GeForce RTX 4090",
|
||||
"gpuTypeId": "NVIDIA L40",
|
||||
"gpuCount": 1,
|
||||
"env": [
|
||||
{
|
||||
@@ -38,6 +38,6 @@
|
||||
"value": "HuggingFaceTB/SmolLM2-135M-Instruct"
|
||||
}
|
||||
],
|
||||
"allowedCudaVersions": ["12.9", "12.8", "12.7", "12.6", "12.5"]
|
||||
"allowedCudaVersions": ["13.0"]
|
||||
}
|
||||
}
|
||||
+9
-6
@@ -1,16 +1,17 @@
|
||||
FROM nvidia/cuda:12.9.1-base-ubuntu22.04
|
||||
FROM nvidia/cuda:13.0.2-devel-ubuntu22.04
|
||||
|
||||
RUN apt-get update -y \
|
||||
&& apt-get install -y python3-pip curl \
|
||||
&& apt-get install -y python3-pip curl git \
|
||||
&& curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||
|
||||
ENV PATH="/root/.local/bin:$PATH"
|
||||
|
||||
RUN ldconfig /usr/local/cuda-12.9/compat/
|
||||
RUN ldconfig /usr/local/cuda-13.0/compat/
|
||||
|
||||
# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels
|
||||
# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
|
||||
RUN uv pip install --system "packaging>=24.2" && \
|
||||
uv pip install --system "vllm[flashinfer]==0.19.1" --extra-index-url https://download.pytorch.org/whl/cu129
|
||||
uv pip install --system "vllm[flashinfer]==0.20.2" && \
|
||||
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation
|
||||
|
||||
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
||||
COPY builder/requirements.txt /requirements.txt
|
||||
@@ -42,7 +43,9 @@ ENV MODEL_NAME=$MODEL_NAME \
|
||||
# Prevent rayon thread pool panic in containers where ulimit -u < nproc
|
||||
# (tokenizers uses Rust's rayon which tries to spawn threads = CPU cores)
|
||||
TOKENIZERS_PARALLELISM=false \
|
||||
RAYON_NUM_THREADS=4
|
||||
RAYON_NUM_THREADS=4 \
|
||||
# Disable DeepGEMM MoE kernels by default; override with VLLM_USE_DEEP_GEMM=1 to enable
|
||||
VLLM_USE_DEEP_GEMM=0
|
||||
|
||||
ENV PYTHONPATH="/:/vllm-workspace"
|
||||
|
||||
|
||||
@@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
|
||||
|
||||

|
||||
|
||||
Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.19.1)
|
||||
Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
|
||||
|
||||
|
||||
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
|
||||
@@ -47,7 +47,7 @@ Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag
|
||||
**📦 Docker Image**: `runpod/worker-v1-vllm:<version>`
|
||||
|
||||
- **Available Versions**: See [GitHub Releases](https://github.com/runpod-workers/worker-vllm/releases)
|
||||
- **CUDA Compatibility**: Requires CUDA >= 12.1
|
||||
- **CUDA Compatibility**: Requires CUDA >= 13.0
|
||||
|
||||
### Configuration
|
||||
|
||||
@@ -78,6 +78,20 @@ Configure worker-vllm using environment variables:
|
||||
|
||||
Any env var whose name matches a valid `AsyncEngineArgs` field (uppercased) is applied automatically. Backward-compat aliases: `MODEL_NAME`, `TOKENIZER_NAME`, `MAX_CONTEXT_LEN_TO_CAPTURE`. This lets you configure any vLLM option without waiting for explicit worker support.
|
||||
|
||||
### Configuration File (config.yaml)
|
||||
|
||||
As an alternative to environment variables, you can supply a `config.yaml` file using the same key names as `vllm serve` (hyphens or underscores both work):
|
||||
|
||||
```yaml
|
||||
model: meta-llama/Llama-3.1-8B-Instruct
|
||||
max-model-len: 8192
|
||||
gpu-memory-utilization: 0.90
|
||||
quantization: awq
|
||||
tensor-parallel-size: 2
|
||||
```
|
||||
|
||||
Mount the file into the container at `/vllm_config.yaml`, or point to a custom path with the `VLLM_CONFIG_FILE` env var. Environment variables always take precedence over config file values.
|
||||
|
||||
For the complete list of all available environment variables, examples, and detailed descriptions: **[Configuration](docs/configuration.md)**
|
||||
|
||||
### Specify Transformers Version
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
ray
|
||||
pandas
|
||||
pyarrow
|
||||
runpod==1.9.0
|
||||
runpod~=1.10.0
|
||||
huggingface-hub
|
||||
lmcache==0.4.2
|
||||
lmcache==0.4.5
|
||||
packaging>=24.2
|
||||
typing-extensions>=4.8.0
|
||||
pydantic
|
||||
@@ -11,5 +11,5 @@ pydantic-settings
|
||||
hf-transfer
|
||||
transformers>=5
|
||||
bitsandbytes>=0.45.0
|
||||
kernels
|
||||
kernels<0.15
|
||||
torch-c-dlpack-ext
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
model: google/gemma-4-31b-it
|
||||
gpu-memory-utilization: 0.95
|
||||
max-model-len: 8192
|
||||
dtype: auto
|
||||
trust-remote-code: true
|
||||
quantization: fp8
|
||||
kv-cache-dtype: fp8
|
||||
enforce-eager: false
|
||||
enable-prefix-caching: true
|
||||
enable-chunked-prefill: true
|
||||
speculative-config: '{"model":"RedHatAI/gemma-4-31B-it-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||
@@ -0,0 +1,9 @@
|
||||
model: openai/gpt-oss-120b
|
||||
gpu-memory-utilization: 0.95
|
||||
max-model-len: 8192
|
||||
dtype: auto
|
||||
trust-remote-code: true
|
||||
enforce-eager: false
|
||||
enable-prefix-caching: true
|
||||
enable-chunked-prefill: true
|
||||
speculative-config: '{"model":"RedHatAI/gpt-oss-120b-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||
@@ -0,0 +1,10 @@
|
||||
model: meta-llama/Llama-3.1-8B-Instruct
|
||||
gpu-memory-utilization: 0.95
|
||||
max-model-len: 8192
|
||||
dtype: auto
|
||||
trust-remote-code: true
|
||||
quantization: fp8
|
||||
kv-cache-dtype: fp8
|
||||
enforce-eager: false
|
||||
enable-prefix-caching: true
|
||||
speculative-config: '{"model":"RedHatAI/Llama-3.1-8B-Instruct-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||
@@ -0,0 +1,10 @@
|
||||
model: Qwen/Qwen3-8B
|
||||
gpu-memory-utilization: 0.95
|
||||
max-model-len: 8192
|
||||
dtype: auto
|
||||
trust-remote-code: true
|
||||
quantization: fp8
|
||||
kv-cache-dtype: fp8
|
||||
enforce-eager: false
|
||||
enable-prefix-caching: true
|
||||
speculative-config: '{"model":"RedHatAI/Qwen3-8B-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||
@@ -97,10 +97,13 @@ If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When
|
||||
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
|
||||
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
|
||||
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. |
|
||||
| `VLLM_USE_DEEP_GEMM` | `0` | `str` (`0`/`1`) | Enable DeepGEMM FP8 kernels for MoE and MQA logits computation. Disabled by default. Must be `"0"` or `"1"` — not `true`/`false`. See note below. |
|
||||
| `ATTENTION_BACKEND` | `None` | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `FLASHINFER`, `TRITON_FLASH_ATTN`). Replaces deprecated `VLLM_ATTENTION_BACKEND`. |
|
||||
| `ASYNC_SCHEDULING` | `None` | `bool` | Enable async scheduling (overlaps engine scheduling with GPU execution). Default: enabled in vLLM 0.14.0+. Set to `false` to disable. |
|
||||
| `STREAM_INTERVAL` | `1` | `int` | Controls how often to yield streaming results. Lower = more frequent updates. |
|
||||
|
||||
> **Note (`VLLM_USE_DEEP_GEMM`):** DeepGEMM is used in two places: MoE weight computation and MQA logits computation. It is necessary for MQA logits computation on supported hardware — required for DeepSeek V4 models. Set `VLLM_USE_DEEP_GEMM=1` to enable. Set `VLLM_USE_DEEP_GEMM=0` to disable the MoE part and fall back to flashinfer/cutlass FP8 kernels. **Value must be `"0"` or `"1"` — not `"true"`/`"false"`.** Some users report better performance with `VLLM_USE_DEEP_GEMM=0`, particularly on H20 GPUs. Disabling it also skips the DeepGEMM warmup phase, reducing cold-start time. Requires CUDA 13.0+ and SM90+ (H100/H200) to use; the library is installed but inactive by default.
|
||||
|
||||
## Tokenizer Settings
|
||||
|
||||
| Variable | Default | Type/Choices | Description |
|
||||
|
||||
+33
-19
@@ -1,4 +1,5 @@
|
||||
import asyncio
|
||||
import inspect
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
@@ -7,6 +8,7 @@ from typing import AsyncGenerator, Optional
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from vllm import AsyncLLMEngine
|
||||
from vllm.inputs import TextPrompt
|
||||
from vllm.entrypoints.logger import RequestLogger
|
||||
from vllm.entrypoints.anthropic.protocol import AnthropicMessagesRequest, AnthropicMessagesResponse, AnthropicError, AnthropicErrorResponse
|
||||
from vllm.entrypoints.anthropic.serving import AnthropicServingMessages
|
||||
@@ -30,20 +32,33 @@ class vLLMEngine:
|
||||
def __init__(self, engine = None):
|
||||
load_dotenv() # For local development
|
||||
self.engine_args = get_engine_args()
|
||||
logging.info(f"Engine args: {self.engine_args}")
|
||||
|
||||
# Initialize vLLM engine first
|
||||
self.llm = self._initialize_llm() if engine is None else engine.llm
|
||||
|
||||
# Only create custom tokenizer wrapper if not using mistral tokenizer mode
|
||||
# For mistral models, let vLLM handle tokenizer initialization
|
||||
if self.engine_args.tokenizer_mode != 'mistral':
|
||||
self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model,
|
||||
self.engine_args.tokenizer_revision,
|
||||
self.engine_args.trust_remote_code)
|
||||
|
||||
if engine is None:
|
||||
ea = self.engine_args
|
||||
summary = {
|
||||
"model": ea.model,
|
||||
"dtype": ea.dtype,
|
||||
"quantization": ea.quantization,
|
||||
"max_model_len": ea.max_model_len,
|
||||
"tensor_parallel_size": ea.tensor_parallel_size,
|
||||
"gpu_memory_utilization": ea.gpu_memory_utilization,
|
||||
}
|
||||
if ea.tokenizer and ea.tokenizer != ea.model:
|
||||
summary["tokenizer"] = ea.tokenizer
|
||||
logging.info("Engine config: %s", summary)
|
||||
logging.debug("Full engine args: %s", ea)
|
||||
|
||||
self.llm = self._initialize_llm()
|
||||
|
||||
if self.engine_args.tokenizer_mode != 'mistral':
|
||||
self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model,
|
||||
self.engine_args.tokenizer_revision,
|
||||
self.engine_args.trust_remote_code)
|
||||
else:
|
||||
self.tokenizer = None
|
||||
else:
|
||||
# For mistral models, we'll get the tokenizer from vLLM later
|
||||
self.tokenizer = None
|
||||
self.llm = engine.llm
|
||||
self.tokenizer = engine.tokenizer
|
||||
|
||||
self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
|
||||
self.default_batch_size = int(os.getenv("DEFAULT_BATCH_SIZE", DEFAULT_BATCH_SIZE))
|
||||
@@ -116,7 +131,7 @@ class vLLMEngine:
|
||||
if apply_chat_template or isinstance(llm_input, list):
|
||||
tokenizer_wrapper = self._get_tokenizer_for_chat_template()
|
||||
llm_input = tokenizer_wrapper.apply_chat_template(llm_input)
|
||||
results_generator = self.llm.generate(llm_input, validated_sampling_params, request_id)
|
||||
results_generator = self.llm.generate(TextPrompt(prompt=llm_input), validated_sampling_params, request_id)
|
||||
n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True
|
||||
last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0}
|
||||
|
||||
@@ -285,7 +300,6 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
self.openai_serving_render = OpenAIServingRender(
|
||||
model_config=self.llm.model_config,
|
||||
renderer=self.llm.renderer,
|
||||
io_processor=self.llm.io_processor,
|
||||
model_registry=self.serving_models.registry,
|
||||
request_logger=None,
|
||||
chat_template=chat_template,
|
||||
@@ -357,10 +371,10 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
||||
)
|
||||
|
||||
if hasattr(self.chat_engine, 'warmup'):
|
||||
import asyncio
|
||||
result = self.chat_engine.warmup()
|
||||
if asyncio.iscoroutine(result):
|
||||
warmup = getattr(self.chat_engine, 'warmup', None)
|
||||
if callable(warmup):
|
||||
result = warmup()
|
||||
if inspect.isawaitable(result):
|
||||
await result
|
||||
|
||||
async def generate(self, openai_request: JobInput):
|
||||
|
||||
+28
-1
@@ -404,6 +404,23 @@ def _resolve_cached_model_path(model_name: str) -> str:
|
||||
return resolved
|
||||
|
||||
|
||||
def _get_args_from_config_file() -> dict:
|
||||
"""Load engine args from a vLLM-style config.yaml.
|
||||
|
||||
Checks VLLM_CONFIG_FILE env var, then falls back to /vllm_config.yaml.
|
||||
Keys use the same long-form names as vllm serve (hyphens converted to underscores).
|
||||
"""
|
||||
import yaml
|
||||
path = os.getenv("VLLM_CONFIG_FILE", "/vllm_config.yaml")
|
||||
if not os.path.exists(path):
|
||||
return {}
|
||||
with open(path) as f:
|
||||
raw = yaml.safe_load(f) or {}
|
||||
normalized = {k.replace("-", "_"): v for k, v in raw.items()}
|
||||
logging.info("Loaded engine args from config file %s: %s", path, list(normalized.keys()))
|
||||
return normalized
|
||||
|
||||
|
||||
def get_local_args():
|
||||
"""
|
||||
Retrieve local arguments from a JSON file.
|
||||
@@ -429,6 +446,9 @@ def get_engine_args():
|
||||
# Start with worker custom defaults (only where we differ from vLLM)
|
||||
args = dict(DEFAULT_ARGS)
|
||||
|
||||
# Config file values sit above defaults but below env vars
|
||||
args.update(_get_args_from_config_file())
|
||||
|
||||
# Auto-discover: every AsyncEngineArgs field from env UPPERCASED (e.g. MAX_MODEL_LEN)
|
||||
args.update(_get_args_from_env_auto_discover())
|
||||
|
||||
@@ -581,6 +601,13 @@ def get_engine_args():
|
||||
|
||||
# Resolve lowercase HF cache paths (FDE-174)
|
||||
if args.get("model"):
|
||||
args["model"] = _resolve_cached_model_path(args["model"])
|
||||
original_model = args["model"]
|
||||
args["model"] = _resolve_cached_model_path(original_model)
|
||||
# When the model was rewritten to an on-disk snapshot path, keep serving
|
||||
# under the original repo id so the OpenAI API model name does not become
|
||||
# a filesystem path (issue #310). An explicit served_model_name (or the
|
||||
# OPENAI_SERVED_MODEL_NAME_OVERRIDE handled downstream) still wins.
|
||||
if args["model"] != original_model and not args.get("served_model_name"):
|
||||
args["served_model_name"] = original_model
|
||||
|
||||
return AsyncEngineArgs(**args)
|
||||
|
||||
+4
-2
@@ -1,10 +1,12 @@
|
||||
from transformers import AutoTokenizer
|
||||
import logging
|
||||
import os
|
||||
from typing import Union
|
||||
|
||||
from transformers import AutoTokenizer
|
||||
|
||||
class TokenizerWrapper:
|
||||
def __init__(self, tokenizer_name_or_path, tokenizer_revision, trust_remote_code):
|
||||
print(f"tokenizer_name_or_path: {tokenizer_name_or_path}, tokenizer_revision: {tokenizer_revision}, trust_remote_code: {trust_remote_code}")
|
||||
logging.debug("tokenizer_name_or_path: %s, tokenizer_revision: %s, trust_remote_code: %s", tokenizer_name_or_path, tokenizer_revision, trust_remote_code)
|
||||
self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_name_or_path, revision=tokenizer_revision or "main", trust_remote_code=trust_remote_code)
|
||||
self.custom_chat_template = os.getenv("CUSTOM_CHAT_TEMPLATE")
|
||||
self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template)
|
||||
|
||||
@@ -0,0 +1,105 @@
|
||||
"""Shared test fixtures.
|
||||
|
||||
``src/engine_args.py`` hard-imports ``vllm`` (and a tensorizer submodule) and
|
||||
``torch.cuda``. Both are only installed inside the GPU Docker image, so when the
|
||||
tests run on a machine without them we install lightweight stubs. When the real
|
||||
packages *are* available (e.g. CI inside the worker image) the stubs are skipped
|
||||
and the real ones are used instead.
|
||||
"""
|
||||
|
||||
import sys
|
||||
import types
|
||||
from dataclasses import dataclass
|
||||
from typing import Optional, Union, List
|
||||
|
||||
|
||||
def _install_torch_stub():
|
||||
try:
|
||||
import torch # noqa: F401
|
||||
return # real torch present, nothing to stub
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
torch = types.ModuleType("torch")
|
||||
cuda = types.ModuleType("torch.cuda")
|
||||
# No GPU in the test environment -> 0 devices (skips tensor-parallel setup).
|
||||
cuda.device_count = lambda: 0
|
||||
torch.cuda = cuda
|
||||
sys.modules["torch"] = torch
|
||||
sys.modules["torch.cuda"] = cuda
|
||||
|
||||
|
||||
def _install_vllm_stub():
|
||||
try:
|
||||
import vllm # noqa: F401
|
||||
return # real vLLM present, nothing to stub
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
vllm = types.ModuleType("vllm")
|
||||
|
||||
@dataclass
|
||||
class AsyncEngineArgs:
|
||||
# Only the fields the worker actually sets/reads need to exist here;
|
||||
# get_engine_args() filters args down to AsyncEngineArgs.__dataclass_fields__
|
||||
# before construction, so unknown keys are dropped rather than passed.
|
||||
model: Optional[str] = None
|
||||
served_model_name: Optional[Union[str, List[str]]] = None
|
||||
revision: Optional[str] = None
|
||||
tokenizer: Optional[str] = None
|
||||
trust_remote_code: bool = False
|
||||
max_model_len: Optional[int] = None
|
||||
max_num_batched_tokens: Optional[int] = None
|
||||
disable_log_stats: bool = False
|
||||
gpu_memory_utilization: float = 0.9
|
||||
tensor_parallel_size: int = 1
|
||||
max_parallel_loading_workers: Optional[int] = None
|
||||
kv_cache_dtype: Optional[str] = None
|
||||
|
||||
class _Stub: # pragma: no cover - placeholder for vllm symbols
|
||||
def __init__(self, *args, **kwargs):
|
||||
pass
|
||||
|
||||
vllm.AsyncEngineArgs = AsyncEngineArgs
|
||||
vllm.SamplingParams = _Stub
|
||||
sys.modules["vllm"] = vllm
|
||||
|
||||
# src.utils imports these at module load and uses ErrorResponse as a return
|
||||
# annotation, which Python evaluates eagerly on <3.14 -> must be defined.
|
||||
vllm_utils = types.ModuleType("vllm.utils")
|
||||
vllm_utils.random_uuid = lambda: "stub-uuid"
|
||||
vllm.utils = vllm_utils
|
||||
sys.modules["vllm.utils"] = vllm_utils
|
||||
|
||||
protocol = types.ModuleType("vllm.entrypoints.openai.engine.protocol")
|
||||
protocol.ErrorResponse = _Stub
|
||||
protocol.ErrorInfo = _Stub
|
||||
protocol.RequestResponseMetadata = _Stub
|
||||
for name in (
|
||||
"vllm.entrypoints",
|
||||
"vllm.entrypoints.openai",
|
||||
"vllm.entrypoints.openai.engine",
|
||||
):
|
||||
sys.modules.setdefault(name, types.ModuleType(name))
|
||||
sys.modules["vllm.entrypoints.openai.engine.protocol"] = protocol
|
||||
|
||||
# vllm.model_executor.model_loader.tensorizer.TensorizerConfig
|
||||
model_executor = types.ModuleType("vllm.model_executor")
|
||||
model_loader = types.ModuleType("vllm.model_executor.model_loader")
|
||||
tensorizer = types.ModuleType("vllm.model_executor.model_loader.tensorizer")
|
||||
|
||||
class TensorizerConfig: # pragma: no cover - placeholder
|
||||
def __init__(self, *args, **kwargs):
|
||||
pass
|
||||
|
||||
tensorizer.TensorizerConfig = TensorizerConfig
|
||||
model_loader.tensorizer = tensorizer
|
||||
model_executor.model_loader = model_loader
|
||||
vllm.model_executor = model_executor
|
||||
sys.modules["vllm.model_executor"] = model_executor
|
||||
sys.modules["vllm.model_executor.model_loader"] = model_loader
|
||||
sys.modules["vllm.model_executor.model_loader.tensorizer"] = tensorizer
|
||||
|
||||
|
||||
_install_torch_stub()
|
||||
_install_vllm_stub()
|
||||
@@ -0,0 +1,6 @@
|
||||
# Test-only dependencies. vllm/torch are stubbed in conftest.py when absent,
|
||||
# so the unit tests run on a plain CPU runner without the GPU image.
|
||||
pytest>=8,<10
|
||||
# get_engine_args() reads a vLLM-style config via PyYAML (a transitive vllm dep
|
||||
# at runtime); install it explicitly here since vllm itself is stubbed.
|
||||
pyyaml
|
||||
@@ -0,0 +1,126 @@
|
||||
"""Tests for HF cache path resolution and served-model-name decoupling.
|
||||
|
||||
Regression coverage for issue #310: when MODEL_NAME is served from a lowercased
|
||||
HF cache dir, the cache resolver rewrites engine_args.model to a snapshot path.
|
||||
The served model name must stay the original repo id, not the path.
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from src import engine_args
|
||||
from src.engine_args import _resolve_cached_model_path, get_engine_args
|
||||
|
||||
|
||||
MODEL = "Qwen/Qwen3.6-27B-FP8"
|
||||
SNAPSHOT_HASH = "e89b16ebf1988b3d6befa7de50abc2d76f26eb09"
|
||||
|
||||
|
||||
def _make_cache(root, folder_name, snapshot=SNAPSHOT_HASH):
|
||||
"""Create a HF-style ``models--…/snapshots/<hash>/`` dir and return its path."""
|
||||
snap_dir = os.path.join(root, folder_name, "snapshots", snapshot)
|
||||
os.makedirs(snap_dir)
|
||||
return snap_dir
|
||||
|
||||
|
||||
def _is_case_sensitive_fs(path):
|
||||
"""The lowercase-cache resolution only matters on case-sensitive filesystems.
|
||||
|
||||
On macOS (APFS, case-insensitive by default) ``models--Qwen--…`` and
|
||||
``models--qwen--…`` collide, so the resolver always sees the exact-case dir
|
||||
as present. Production runs on Linux (case-sensitive), which is what these
|
||||
tests exercise.
|
||||
"""
|
||||
probe = os.path.join(path, "CaseProbe")
|
||||
open(probe, "w").close()
|
||||
try:
|
||||
return not os.path.exists(os.path.join(path, "caseprobe"))
|
||||
finally:
|
||||
os.remove(probe)
|
||||
|
||||
|
||||
requires_case_sensitive_fs = pytest.mark.skipif(
|
||||
not _is_case_sensitive_fs(os.environ.get("TMPDIR", "/tmp")),
|
||||
reason="lowercase HF cache resolution only applies on case-sensitive filesystems",
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def hf_cache(tmp_path, monkeypatch):
|
||||
cache = tmp_path / "hub"
|
||||
cache.mkdir()
|
||||
monkeypatch.setenv("HUGGINGFACE_HUB_CACHE", str(cache))
|
||||
# Make sure HF_HOME does not shadow the explicit cache dir during the test.
|
||||
monkeypatch.delenv("HF_HOME", raising=False)
|
||||
return cache
|
||||
|
||||
|
||||
class TestResolveCachedModelPath:
|
||||
def test_exact_case_dir_returns_repo_id(self, hf_cache):
|
||||
_make_cache(str(hf_cache), "models--Qwen--Qwen3.6-27B-FP8")
|
||||
assert _resolve_cached_model_path(MODEL) == MODEL
|
||||
|
||||
def test_no_cache_returns_repo_id(self, hf_cache):
|
||||
assert _resolve_cached_model_path(MODEL) == MODEL
|
||||
|
||||
def test_absolute_path_passthrough(self, hf_cache):
|
||||
path = "/runpod-volume/some/local/model"
|
||||
assert _resolve_cached_model_path(path) == path
|
||||
|
||||
@requires_case_sensitive_fs
|
||||
def test_lowercase_dir_returns_snapshot_path(self, hf_cache):
|
||||
snap = _make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")
|
||||
assert _resolve_cached_model_path(MODEL) == snap
|
||||
|
||||
def test_lowercase_dir_without_snapshots_returns_repo_id(self, hf_cache):
|
||||
# Dir exists but has no snapshots subdir -> nothing to resolve to.
|
||||
os.makedirs(os.path.join(str(hf_cache), "models--qwen--qwen3.6-27b-fp8"))
|
||||
assert _resolve_cached_model_path(MODEL) == MODEL
|
||||
|
||||
@requires_case_sensitive_fs
|
||||
def test_lowercase_dir_picks_latest_snapshot(self, hf_cache):
|
||||
folder = "models--qwen--qwen3.6-27b-fp8"
|
||||
_make_cache(str(hf_cache), folder, snapshot="aaaa")
|
||||
latest = _make_cache(str(hf_cache), folder, snapshot="zzzz")
|
||||
assert _resolve_cached_model_path(MODEL) == latest
|
||||
|
||||
|
||||
class TestGetEngineArgsServedName:
|
||||
"""Issue #310: served name must be decoupled from the resolved on-disk path."""
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def base_env(self, monkeypatch):
|
||||
# Avoid the network branch in _resolve_max_model_len.
|
||||
monkeypatch.setenv("MAX_NUM_BATCHED_TOKENS", "2048")
|
||||
monkeypatch.delenv("SERVED_MODEL_NAME", raising=False)
|
||||
# Don't pick up a stray vLLM config file from the environment.
|
||||
monkeypatch.setenv("VLLM_CONFIG_FILE", "/nonexistent-vllm-config.yaml")
|
||||
|
||||
@requires_case_sensitive_fs
|
||||
def test_served_name_is_repo_id_when_path_rewritten(self, hf_cache, monkeypatch):
|
||||
snap = _make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")
|
||||
monkeypatch.setenv("MODEL_NAME", MODEL)
|
||||
|
||||
result = get_engine_args()
|
||||
|
||||
assert result.model == snap # weights load from the lowercase cache
|
||||
assert result.served_model_name == MODEL # API still serves the repo id
|
||||
|
||||
def test_served_name_untouched_when_no_rewrite(self, hf_cache, monkeypatch):
|
||||
_make_cache(str(hf_cache), "models--Qwen--Qwen3.6-27B-FP8")
|
||||
monkeypatch.setenv("MODEL_NAME", MODEL)
|
||||
|
||||
result = get_engine_args()
|
||||
|
||||
assert result.model == MODEL
|
||||
assert result.served_model_name is None
|
||||
|
||||
def test_explicit_served_name_not_overridden(self, hf_cache, monkeypatch):
|
||||
_make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")
|
||||
monkeypatch.setenv("MODEL_NAME", MODEL)
|
||||
monkeypatch.setenv("SERVED_MODEL_NAME", "custom-name")
|
||||
|
||||
result = get_engine_args()
|
||||
|
||||
assert result.served_model_name == "custom-name"
|
||||
Reference in New Issue
Block a user