Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
76f345be08 | ||
|
|
4dfda80fd0 | ||
|
|
89f25b0d70 | ||
|
|
4808160735 | ||
|
|
51e1aee770 | ||
|
|
86b7951c39 | ||
|
|
e786204cf5 | ||
|
|
591f9d6531 | ||
|
|
49967d0c09 | ||
|
|
e97cab854f | ||
|
|
828e797498 | ||
|
|
373847d9f9 | ||
|
|
2000acc0a3 | ||
|
|
89e3b2e920 | ||
|
|
85f8c81c1c | ||
|
|
88490e9551 | ||
|
|
798f2b12eb | ||
|
|
9e1c483136 | ||
|
|
d71ea9939d | ||
|
|
7e2b4e2288 | ||
|
|
015f8f3c4d | ||
|
|
75ffcf73f2 | ||
|
|
d7ba3b6ab7 | ||
|
|
b11c91722c | ||
|
|
fcdc799e0d | ||
|
|
84ec446493 | ||
|
|
db246653a2 | ||
|
|
1b3228a2dc | ||
|
|
4817d4a8e7 | ||
|
|
0378382a92 | ||
|
|
08580e7ccf | ||
|
|
8868aae6b1 | ||
|
|
fb8adc5c06 | ||
|
|
7351da512b | ||
|
|
1e78043b2c | ||
|
|
352c64f4c1 | ||
|
|
0922f5b435 | ||
|
|
5d9a48fc70 | ||
|
|
5d1579e361 | ||
|
|
0488b77d89 | ||
|
|
d3a962c33b | ||
|
|
105c125698 | ||
|
|
0a0ccfcb60 | ||
|
|
c8ce53c72c | ||
|
|
9618e799ba | ||
|
|
cb3f077dba | ||
|
|
69646b9e99 | ||
|
|
8b991a7ad7 | ||
|
|
d356c31675 | ||
|
|
80072047ab |
@@ -0,0 +1,71 @@
|
|||||||
|
name: CI | Sync vLLM version in READMEs
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches: ["main"]
|
||||||
|
paths:
|
||||||
|
- "Dockerfile"
|
||||||
|
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: write
|
||||||
|
pull-requests: write
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
sync_version:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
name: Check README version matches Dockerfile and update if needed
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Extract vLLM version from Dockerfile and sync READMEs
|
||||||
|
run: |
|
||||||
|
echo "Extracting vLLM version from Dockerfile..."
|
||||||
|
dockerfile_version=$(grep -oP 'vllm(?:\[[\w,]+\])?==\K[\d.]+' Dockerfile | head -1)
|
||||||
|
|
||||||
|
if [ -z "$dockerfile_version" ]; then
|
||||||
|
echo "ERROR: Could not extract vLLM version from Dockerfile."
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
echo "Dockerfile vLLM version: $dockerfile_version"
|
||||||
|
echo "VLLM_VERSION=$dockerfile_version" >> $GITHUB_ENV
|
||||||
|
|
||||||
|
updated=0
|
||||||
|
for readme in README.md .runpod/README.md; do
|
||||||
|
if [ ! -f "$readme" ]; then
|
||||||
|
echo "Skipping $readme (not found)"
|
||||||
|
continue
|
||||||
|
fi
|
||||||
|
|
||||||
|
readme_version=$(grep -oP 'Current vLLM version: \[\K[\d.]+' "$readme" || echo "")
|
||||||
|
echo "$readme current version: ${readme_version:-not found}"
|
||||||
|
|
||||||
|
if [ "$readme_version" = "$dockerfile_version" ]; then
|
||||||
|
echo "$readme is already up to date."
|
||||||
|
continue
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "Updating $readme from $readme_version to $dockerfile_version..."
|
||||||
|
sed -i "s|Current vLLM version: \[${readme_version}\](https://github.com/vllm-project/vllm/releases/tag/v${readme_version})|Current vLLM version: [${dockerfile_version}](https://github.com/vllm-project/vllm/releases/tag/v${dockerfile_version})|g" "$readme"
|
||||||
|
updated=1
|
||||||
|
done
|
||||||
|
|
||||||
|
echo "UPDATED=$updated" >> $GITHUB_ENV
|
||||||
|
|
||||||
|
- name: Create Pull Request
|
||||||
|
if: env.UPDATED == '1'
|
||||||
|
uses: peter-evans/create-pull-request@v7
|
||||||
|
with:
|
||||||
|
token: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
commit-message: "docs: sync vLLM version to ${{ env.VLLM_VERSION }} in READMEs"
|
||||||
|
title: "docs: sync vLLM version to ${{ env.VLLM_VERSION }} in READMEs"
|
||||||
|
body: |
|
||||||
|
The vLLM version in the Dockerfile has been updated to `${{ env.VLLM_VERSION }}`.
|
||||||
|
|
||||||
|
This PR syncs the version badge/link in:
|
||||||
|
- `README.md`
|
||||||
|
- `.runpod/README.md`
|
||||||
|
branch: docs/sync-vllm-version-${{ env.VLLM_VERSION }}
|
||||||
|
labels: documentation
|
||||||
@@ -0,0 +1,162 @@
|
|||||||
|
name: Serverless Model Tests
|
||||||
|
|
||||||
|
on:
|
||||||
|
pull_request:
|
||||||
|
branches:
|
||||||
|
- "**"
|
||||||
|
push:
|
||||||
|
branches:
|
||||||
|
- "main"
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
|
||||||
|
# Only one run per ref at a time — these tests spin up real H100 endpoints.
|
||||||
|
concurrency:
|
||||||
|
group: serverless-model-tests-${{ github.ref }}
|
||||||
|
cancel-in-progress: true
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
# On PRs we only smoke-test one (small, cheap) model to keep review loops fast.
|
||||||
|
test-pr:
|
||||||
|
if: github.event_name == 'pull_request'
|
||||||
|
runs-on: [blacksmith-8vcpu-ubuntu-2204, linux]
|
||||||
|
timeout-minutes: 60
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Set up QEMU
|
||||||
|
uses: docker/setup-qemu-action@v3
|
||||||
|
|
||||||
|
- name: Set up Docker Buildx
|
||||||
|
uses: docker/setup-buildx-action@v3
|
||||||
|
|
||||||
|
- name: Login to Docker Hub
|
||||||
|
uses: docker/login-action@v3
|
||||||
|
with:
|
||||||
|
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||||
|
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||||
|
|
||||||
|
- name: Set up Python
|
||||||
|
uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: "3.11"
|
||||||
|
|
||||||
|
- name: Install script dependencies
|
||||||
|
run: pip install pyyaml
|
||||||
|
|
||||||
|
- name: Run serverless e2e test
|
||||||
|
id: e2e_test
|
||||||
|
run: python scripts/serverless_e2e_test.py --config configs/gpt-oss/gpt_oss_120b.yaml --build
|
||||||
|
env:
|
||||||
|
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY_2 }}
|
||||||
|
DOCKERHUB_REPO: ${{ vars.DOCKERHUB_REPO || 'runpod' }}
|
||||||
|
DOCKERHUB_IMG: ${{ vars.DOCKERHUB_IMG || 'worker-v1-vllm' }}
|
||||||
|
HUGGINGFACE_ACCESS_TOKEN: ${{ secrets.HUGGINGFACE_ACCESS_TOKEN_2 }}
|
||||||
|
|
||||||
|
- name: Cleanup safety net
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
if [ -n "${{ steps.e2e_test.outputs.endpoint_id }}" ]; then
|
||||||
|
curl -sf -X DELETE "https://rest.runpod.io/v1/endpoints/${{ steps.e2e_test.outputs.endpoint_id }}" \
|
||||||
|
-H "Authorization: Bearer ${{ secrets.RUNPOD_API_KEY_2 }}" || true
|
||||||
|
fi
|
||||||
|
if [ -n "${{ steps.e2e_test.outputs.template_id }}" ]; then
|
||||||
|
curl -sf -X DELETE "https://rest.runpod.io/v1/templates/${{ steps.e2e_test.outputs.template_id }}" \
|
||||||
|
-H "Authorization: Bearer ${{ secrets.RUNPOD_API_KEY_2 }}" || true
|
||||||
|
fi
|
||||||
|
|
||||||
|
# On push to main we test every tuned model config in parallel.
|
||||||
|
discover:
|
||||||
|
if: github.event_name == 'push'
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
outputs:
|
||||||
|
configs: ${{ steps.list.outputs.configs }}
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: List model configs
|
||||||
|
id: list
|
||||||
|
run: |
|
||||||
|
CONFIGS=$(find configs -name '*.yaml' | sort | jq -R -s -c 'split("\n") | map(select(length > 0))')
|
||||||
|
echo "configs=$CONFIGS" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
|
# The image is the same regardless of which model config is under test (configs
|
||||||
|
# only supply endpoint env vars), so build/push it once and let every matrix job
|
||||||
|
# in test-main reuse that same tag instead of rebuilding per config.
|
||||||
|
build:
|
||||||
|
if: github.event_name == 'push'
|
||||||
|
runs-on: [blacksmith-8vcpu-ubuntu-2204, linux]
|
||||||
|
outputs:
|
||||||
|
release_version: ${{ steps.build.outputs.release_version }}
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Set up QEMU
|
||||||
|
uses: docker/setup-qemu-action@v3
|
||||||
|
|
||||||
|
- name: Set up Docker Buildx
|
||||||
|
uses: docker/setup-buildx-action@v3
|
||||||
|
|
||||||
|
- name: Login to Docker Hub
|
||||||
|
uses: docker/login-action@v3
|
||||||
|
with:
|
||||||
|
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||||
|
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||||
|
|
||||||
|
- name: Build and push image
|
||||||
|
id: build
|
||||||
|
env:
|
||||||
|
DOCKERHUB_REPO: ${{ vars.DOCKERHUB_REPO || 'runpod' }}
|
||||||
|
DOCKERHUB_IMG: ${{ vars.DOCKERHUB_IMG || 'worker-v1-vllm' }}
|
||||||
|
RELEASE_VERSION: test-${{ github.sha }}
|
||||||
|
HUGGINGFACE_ACCESS_TOKEN: ${{ secrets.HUGGINGFACE_ACCESS_TOKEN }}
|
||||||
|
run: |
|
||||||
|
docker buildx bake --push
|
||||||
|
echo "release_version=${RELEASE_VERSION}" >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
|
test-main:
|
||||||
|
needs: [discover, build]
|
||||||
|
if: github.event_name == 'push' && needs.discover.result == 'success' && needs.build.result == 'success'
|
||||||
|
runs-on: [blacksmith-8vcpu-ubuntu-2204, linux]
|
||||||
|
timeout-minutes: 60
|
||||||
|
strategy:
|
||||||
|
fail-fast: false
|
||||||
|
matrix:
|
||||||
|
config: ${{ fromJSON(needs.discover.outputs.configs) }}
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Set up Python
|
||||||
|
uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: "3.11"
|
||||||
|
|
||||||
|
- name: Install script dependencies
|
||||||
|
run: pip install pyyaml
|
||||||
|
|
||||||
|
- name: Run serverless e2e test
|
||||||
|
id: e2e_test
|
||||||
|
run: python scripts/serverless_e2e_test.py --config "${{ matrix.config }}" --image "${DOCKERHUB_REPO}/${DOCKERHUB_IMG}:${RELEASE_VERSION}"
|
||||||
|
env:
|
||||||
|
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY_2 }}
|
||||||
|
HUGGINGFACE_ACCESS_TOKEN: ${{ secrets.HUGGINGFACE_ACCESS_TOKEN_2 }}
|
||||||
|
DOCKERHUB_REPO: ${{ vars.DOCKERHUB_REPO || 'runpod' }}
|
||||||
|
DOCKERHUB_IMG: ${{ vars.DOCKERHUB_IMG || 'worker-v1-vllm' }}
|
||||||
|
RELEASE_VERSION: ${{ needs.build.outputs.release_version }}
|
||||||
|
|
||||||
|
- name: Cleanup safety net
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
if [ -n "${{ steps.e2e_test.outputs.endpoint_id }}" ]; then
|
||||||
|
curl -sf -X DELETE "https://rest.runpod.io/v1/endpoints/${{ steps.e2e_test.outputs.endpoint_id }}" \
|
||||||
|
-H "Authorization: Bearer ${{ secrets.RUNPOD_API_KEY_2 }}" || true
|
||||||
|
fi
|
||||||
|
if [ -n "${{ steps.e2e_test.outputs.template_id }}" ]; then
|
||||||
|
curl -sf -X DELETE "https://rest.runpod.io/v1/templates/${{ steps.e2e_test.outputs.template_id }}" \
|
||||||
|
-H "Authorization: Bearer ${{ secrets.RUNPOD_API_KEY_2 }}" || true
|
||||||
|
fi
|
||||||
@@ -0,0 +1,32 @@
|
|||||||
|
name: Tests
|
||||||
|
|
||||||
|
on:
|
||||||
|
pull_request:
|
||||||
|
branches:
|
||||||
|
- "**"
|
||||||
|
push:
|
||||||
|
branches:
|
||||||
|
- "main"
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
pytest:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Set up Python
|
||||||
|
uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: "3.11"
|
||||||
|
|
||||||
|
- name: Install test dependencies
|
||||||
|
run: |
|
||||||
|
python -m pip install --upgrade pip
|
||||||
|
pip install -r tests/requirements.txt
|
||||||
|
|
||||||
|
- name: Run unit tests
|
||||||
|
run: python -m pytest tests -v
|
||||||
+12
-1
@@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
|
|||||||
|
|
||||||
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
||||||
|
|
||||||
Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
|
Current vLLM version: [0.23.0](https://github.com/vllm-project/vllm/releases/tag/v0.23.0)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -33,6 +33,17 @@ All behaviour is controlled through environment variables:
|
|||||||
|
|
||||||
**Pass any vLLM engine arg** not listed above by setting an env var with the **UPPERCASED** field name (e.g. `MAX_MODEL_LEN=4096`, `ENABLE_CHUNKED_PREFILL=true`). The worker auto-discovers all `AsyncEngineArgs` fields from env. See the [vLLM engine args docs](https://docs.vllm.ai/en/latest/configuration/engine_args) for all available options.
|
**Pass any vLLM engine arg** not listed above by setting an env var with the **UPPERCASED** field name (e.g. `MAX_MODEL_LEN=4096`, `ENABLE_CHUNKED_PREFILL=true`). The worker auto-discovers all `AsyncEngineArgs` fields from env. See the [vLLM engine args docs](https://docs.vllm.ai/en/latest/configuration/engine_args) for all available options.
|
||||||
|
|
||||||
|
**Configuration file:** You can also supply a `config.yaml` instead of (or alongside) env vars. Mount it at `/vllm_config.yaml` in the container, or set `VLLM_CONFIG_FILE` to a custom path. Use the same key names as `vllm serve` — hyphens and underscores both work:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model: meta-llama/Llama-3.1-8B-Instruct
|
||||||
|
max-model-len: 8192
|
||||||
|
gpu-memory-utilization: 0.90
|
||||||
|
quantization: awq
|
||||||
|
```
|
||||||
|
|
||||||
|
Environment variables always override config file values.
|
||||||
|
|
||||||
For complete configuration options, see the [full configuration documentation](https://github.com/runpod-workers/worker-vllm/blob/main/docs/configuration.md).
|
For complete configuration options, see the [full configuration documentation](https://github.com/runpod-workers/worker-vllm/blob/main/docs/configuration.md).
|
||||||
|
|
||||||
### Specify Transformers Version
|
### Specify Transformers Version
|
||||||
|
|||||||
@@ -247,6 +247,7 @@
|
|||||||
"name": "Max Parallel Loading Workers",
|
"name": "Max Parallel Loading Workers",
|
||||||
"type": "number",
|
"type": "number",
|
||||||
"description": "Load model sequentially in multiple batches.",
|
"description": "Load model sequentially in multiple batches.",
|
||||||
|
"default": 1,
|
||||||
"advanced": true
|
"advanced": true
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
|
|||||||
+1
-1
@@ -30,7 +30,7 @@
|
|||||||
}
|
}
|
||||||
],
|
],
|
||||||
"config": {
|
"config": {
|
||||||
"gpuTypeId": "NVIDIA GeForce RTX 4090",
|
"gpuTypeId": "NVIDIA L40",
|
||||||
"gpuCount": 1,
|
"gpuCount": 1,
|
||||||
"env": [
|
"env": [
|
||||||
{
|
{
|
||||||
|
|||||||
+26
-6
@@ -1,17 +1,36 @@
|
|||||||
FROM nvidia/cuda:13.0.2-devel-ubuntu22.04
|
FROM nvidia/cuda:13.0.2-devel-ubuntu22.04
|
||||||
|
|
||||||
|
ENV DEBIAN_FRONTEND=noninteractive
|
||||||
|
|
||||||
RUN apt-get update -y \
|
RUN apt-get update -y \
|
||||||
&& apt-get install -y python3-pip curl git \
|
&& apt-get install -y curl git software-properties-common \
|
||||||
&& curl -LsSf https://astral.sh/uv/install.sh | sh
|
&& add-apt-repository -y ppa:deadsnakes/ppa \
|
||||||
|
&& apt-get install -y python3.12 python3.12-dev python3.12-venv \
|
||||||
|
&& update-alternatives --install /usr/bin/python3 python3 /usr/bin/python3.12 1 \
|
||||||
|
&& update-alternatives --set python3 /usr/bin/python3.12 \
|
||||||
|
&& rm -f /usr/lib/python3.12/EXTERNALLY-MANAGED \
|
||||||
|
&& curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||||
|
|
||||||
ENV PATH="/root/.local/bin:$PATH"
|
ENV PATH="/root/.local/bin:$PATH"
|
||||||
|
|
||||||
RUN ldconfig /usr/local/cuda-13.0/compat/
|
RUN ldconfig /usr/local/cuda-13.0/compat/
|
||||||
|
|
||||||
# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
|
# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
|
||||||
RUN uv pip install --system "packaging>=24.2" && \
|
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||||
uv pip install --system "vllm[flashinfer]==0.20.2" && \
|
uv pip install --system "packaging>=24.2" && \
|
||||||
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation
|
uv pip install --system "vllm[flashinfer]==0.23.0" && \
|
||||||
|
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation && \
|
||||||
|
uv pip install --system --force-reinstall --no-deps nixl-cu13
|
||||||
|
|
||||||
|
# Fix CUTLASS DSL cu13 install order: nvidia-cutlass-dsl[cu13] installs
|
||||||
|
# -libs-base and -libs-cu13 wheels that share paths with different content.
|
||||||
|
# uv can extract them in either order, leaving base files that break CUDA 13
|
||||||
|
# CuTe DSL JIT. Force -libs-cu13 last. See vllm-project/vllm#45204.
|
||||||
|
RUN CUTLASS_DSL_VERSION=$(uv pip show --system nvidia-cutlass-dsl 2>/dev/null | awk '/^Version:/{print $2}') && \
|
||||||
|
if [ -n "$CUTLASS_DSL_VERSION" ]; then \
|
||||||
|
uv pip install --system --force-reinstall --no-deps \
|
||||||
|
"nvidia-cutlass-dsl-libs-cu13==${CUTLASS_DSL_VERSION}"; \
|
||||||
|
fi
|
||||||
|
|
||||||
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
||||||
COPY builder/requirements.txt /requirements.txt
|
COPY builder/requirements.txt /requirements.txt
|
||||||
@@ -47,7 +66,8 @@ ENV MODEL_NAME=$MODEL_NAME \
|
|||||||
# Disable DeepGEMM MoE kernels by default; override with VLLM_USE_DEEP_GEMM=1 to enable
|
# Disable DeepGEMM MoE kernels by default; override with VLLM_USE_DEEP_GEMM=1 to enable
|
||||||
VLLM_USE_DEEP_GEMM=0
|
VLLM_USE_DEEP_GEMM=0
|
||||||
|
|
||||||
ENV PYTHONPATH="/:/vllm-workspace"
|
ENV PYTHONPATH="/:/vllm-workspace" \
|
||||||
|
LD_LIBRARY_PATH="/usr/local/nvidia/lib64:/usr/local/cuda/lib64:${LD_LIBRARY_PATH}"
|
||||||
|
|
||||||
RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \
|
RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \
|
||||||
uv pip install --system -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \
|
uv pip install --system -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
|
|||||||
|
|
||||||

|

|
||||||
|
|
||||||
Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
|
Current vLLM version: [0.23.0](https://github.com/vllm-project/vllm/releases/tag/v0.23.0)
|
||||||
|
|
||||||
|
|
||||||
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
|
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
|
||||||
@@ -78,6 +78,20 @@ Configure worker-vllm using environment variables:
|
|||||||
|
|
||||||
Any env var whose name matches a valid `AsyncEngineArgs` field (uppercased) is applied automatically. Backward-compat aliases: `MODEL_NAME`, `TOKENIZER_NAME`, `MAX_CONTEXT_LEN_TO_CAPTURE`. This lets you configure any vLLM option without waiting for explicit worker support.
|
Any env var whose name matches a valid `AsyncEngineArgs` field (uppercased) is applied automatically. Backward-compat aliases: `MODEL_NAME`, `TOKENIZER_NAME`, `MAX_CONTEXT_LEN_TO_CAPTURE`. This lets you configure any vLLM option without waiting for explicit worker support.
|
||||||
|
|
||||||
|
### Configuration File (config.yaml)
|
||||||
|
|
||||||
|
As an alternative to environment variables, you can supply a `config.yaml` file using the same key names as `vllm serve` (hyphens or underscores both work):
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model: meta-llama/Llama-3.1-8B-Instruct
|
||||||
|
max-model-len: 8192
|
||||||
|
gpu-memory-utilization: 0.90
|
||||||
|
quantization: awq
|
||||||
|
tensor-parallel-size: 2
|
||||||
|
```
|
||||||
|
|
||||||
|
Mount the file into the container at `/vllm_config.yaml`, or point to a custom path with the `VLLM_CONFIG_FILE` env var. Environment variables always take precedence over config file values.
|
||||||
|
|
||||||
For the complete list of all available environment variables, examples, and detailed descriptions: **[Configuration](docs/configuration.md)**
|
For the complete list of all available environment variables, examples, and detailed descriptions: **[Configuration](docs/configuration.md)**
|
||||||
|
|
||||||
### Specify Transformers Version
|
### Specify Transformers Version
|
||||||
|
|||||||
@@ -1,9 +1,9 @@
|
|||||||
ray
|
ray
|
||||||
pandas
|
pandas
|
||||||
pyarrow
|
pyarrow
|
||||||
runpod==1.9.0
|
runpod~=1.11.0
|
||||||
huggingface-hub
|
huggingface-hub
|
||||||
lmcache==0.4.5
|
lmcache==0.5.0
|
||||||
packaging>=24.2
|
packaging>=24.2
|
||||||
typing-extensions>=4.8.0
|
typing-extensions>=4.8.0
|
||||||
pydantic
|
pydantic
|
||||||
|
|||||||
@@ -0,0 +1,12 @@
|
|||||||
|
model: google/gemma-4-31b-it
|
||||||
|
gpu-memory-utilization: 0.95
|
||||||
|
max-model-len: 8192
|
||||||
|
dtype: auto
|
||||||
|
trust-remote-code: true
|
||||||
|
quantization: fp8
|
||||||
|
kv-cache-dtype: fp8
|
||||||
|
enforce-eager: false
|
||||||
|
enable-prefix-caching: true
|
||||||
|
enable-chunked-prefill: true
|
||||||
|
vllm-release: v2.22.5
|
||||||
|
speculative-config: '{"model":"RedHatAI/gemma-4-31B-it-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||||
@@ -0,0 +1,9 @@
|
|||||||
|
model: openai/gpt-oss-120b
|
||||||
|
gpu-memory-utilization: 0.95
|
||||||
|
max-model-len: 8192
|
||||||
|
dtype: auto
|
||||||
|
trust-remote-code: true
|
||||||
|
enforce-eager: false
|
||||||
|
enable-prefix-caching: true
|
||||||
|
enable-chunked-prefill: true
|
||||||
|
speculative-config: '{"model":"RedHatAI/gpt-oss-120b-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
model: meta-llama/Llama-3.1-8B-Instruct
|
||||||
|
gpu-memory-utilization: 0.95
|
||||||
|
max-model-len: 8192
|
||||||
|
dtype: auto
|
||||||
|
trust-remote-code: true
|
||||||
|
quantization: fp8
|
||||||
|
kv-cache-dtype: fp8
|
||||||
|
enforce-eager: false
|
||||||
|
vllm-release: v2.22.5
|
||||||
|
enable-prefix-caching: true
|
||||||
|
speculative-config: '{"model":"RedHatAI/Llama-3.1-8B-Instruct-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||||
@@ -0,0 +1,12 @@
|
|||||||
|
model: Qwen/Qwen3-8B
|
||||||
|
gpu-memory-utilization: 0.95
|
||||||
|
max-model-len: 8192
|
||||||
|
dtype: auto
|
||||||
|
trust-remote-code: true
|
||||||
|
quantization: fp8
|
||||||
|
kv-cache-dtype: fp8
|
||||||
|
enforce-eager: false
|
||||||
|
enable-prefix-caching: true
|
||||||
|
vllm-release: v2.22.5
|
||||||
|
compilation-config: '{"cudagraph_mode": "PIECEWISE"}'
|
||||||
|
speculative-config: '{"model":"RedHatAI/Qwen3-8B-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||||
@@ -0,0 +1,328 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""End-to-end test: build/push the worker image, deploy it as a real RunPod
|
||||||
|
Serverless endpoint on H100, run the .runpod/tests.json test cases against it,
|
||||||
|
then tear the endpoint down.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
RUNPOD_API_KEY=... python scripts/serverless_e2e_test.py \\
|
||||||
|
--config configs/qwen/qwen3_8b.yaml --build
|
||||||
|
|
||||||
|
RUNPOD_API_KEY=... python scripts/serverless_e2e_test.py \\
|
||||||
|
--config configs/qwen/qwen3_8b.yaml --image runpod/worker-v1-vllm:dev-my-branch
|
||||||
|
"""
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
import urllib.error
|
||||||
|
import urllib.request
|
||||||
|
import uuid
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import yaml
|
||||||
|
|
||||||
|
REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||||
|
|
||||||
|
REST_API_BASE = "https://rest.runpod.io/v1"
|
||||||
|
JOB_API_BASE = "https://api.runpod.ai/v2"
|
||||||
|
|
||||||
|
DEFAULT_GPU_TYPE_IDS = [
|
||||||
|
"NVIDIA H100 80GB HBM3",
|
||||||
|
"NVIDIA H100 NVL",
|
||||||
|
"NVIDIA H100 PCIe",
|
||||||
|
]
|
||||||
|
|
||||||
|
TERMINAL_STATUSES = {"COMPLETED", "FAILED", "TIMED_OUT", "CANCELLED"}
|
||||||
|
|
||||||
|
|
||||||
|
def log(msg: str) -> None:
|
||||||
|
print(f"[serverless_e2e_test] {msg}", flush=True)
|
||||||
|
|
||||||
|
|
||||||
|
def write_github_output(key: str, value: str) -> None:
|
||||||
|
"""Expose a value to later CI steps (e.g. an always() cleanup safety net
|
||||||
|
for when this process gets killed before its own `finally` can run)."""
|
||||||
|
path = os.environ.get("GITHUB_OUTPUT")
|
||||||
|
if not path:
|
||||||
|
return
|
||||||
|
with open(path, "a") as f:
|
||||||
|
f.write(f"{key}={value}\n")
|
||||||
|
|
||||||
|
|
||||||
|
def stringify_env_value(value) -> str:
|
||||||
|
if isinstance(value, bool):
|
||||||
|
return "true" if value else "false"
|
||||||
|
if isinstance(value, (dict, list)):
|
||||||
|
return json.dumps(value)
|
||||||
|
return str(value)
|
||||||
|
|
||||||
|
|
||||||
|
def load_hub_defaults(hub_json_path: Path) -> dict:
|
||||||
|
hub = json.loads(hub_json_path.read_text())
|
||||||
|
config = hub.get("config", {})
|
||||||
|
|
||||||
|
base_env = {}
|
||||||
|
for entry in config.get("env", []):
|
||||||
|
default = entry.get("input", {}).get("default")
|
||||||
|
if default is None:
|
||||||
|
continue
|
||||||
|
base_env[entry["key"]] = stringify_env_value(default)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"containerDiskInGb": config.get("containerDiskInGb", 50),
|
||||||
|
"gpuCount": config.get("gpuCount", 1),
|
||||||
|
"allowedCudaVersions": config.get("allowedCudaVersions"),
|
||||||
|
"minCudaVersion": config.get("minCudaVersion"),
|
||||||
|
"env": base_env,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
# config keys whose naive `KEY.replace("-", "_").upper()` transform doesn't match
|
||||||
|
# the env var the worker actually reads (see src/engine_args.py ENV_ALIASES).
|
||||||
|
CONFIG_KEY_ALIASES = {
|
||||||
|
"MODEL": "MODEL_NAME",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def load_model_env(config_yaml_path: Path) -> dict:
|
||||||
|
raw = yaml.safe_load(config_yaml_path.read_text()) or {}
|
||||||
|
env = {}
|
||||||
|
for key, value in raw.items():
|
||||||
|
env_key = key.replace("-", "_").upper()
|
||||||
|
env_key = CONFIG_KEY_ALIASES.get(env_key, env_key)
|
||||||
|
env[env_key] = stringify_env_value(value)
|
||||||
|
return env
|
||||||
|
|
||||||
|
|
||||||
|
def build_and_push_image(dockerhub_repo: str, dockerhub_img: str, release_version: str) -> str:
|
||||||
|
hf_token = os.environ.get("HUGGINGFACE_ACCESS_TOKEN", "")
|
||||||
|
dockerhub_user = os.environ.get("DOCKERHUB_USERNAME")
|
||||||
|
dockerhub_pass = os.environ.get("DOCKERHUB_TOKEN")
|
||||||
|
|
||||||
|
if dockerhub_user and dockerhub_pass:
|
||||||
|
log(f"Logging in to Docker Hub as {dockerhub_user}")
|
||||||
|
subprocess.run(
|
||||||
|
["docker", "login", "-u", dockerhub_user, "--password-stdin"],
|
||||||
|
input=dockerhub_pass,
|
||||||
|
text=True,
|
||||||
|
check=True,
|
||||||
|
cwd=REPO_ROOT,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
log("DOCKERHUB_USERNAME/DOCKERHUB_TOKEN not set; assuming docker is already logged in")
|
||||||
|
|
||||||
|
tag = f"{dockerhub_repo}/{dockerhub_img}:{release_version}"
|
||||||
|
log(f"Building and pushing {tag} via docker buildx bake")
|
||||||
|
# docker-bake.hcl's DOCKERHUB_REPO/DOCKERHUB_IMG/RELEASE_VERSION are bake-level
|
||||||
|
# `variable` blocks that compute `tags` - they read from env vars of the same
|
||||||
|
# name, NOT from `--set target.args.*` (that sets Dockerfile build ARGs, a
|
||||||
|
# separate namespace the Dockerfile doesn't even declare these under).
|
||||||
|
bake_env = {
|
||||||
|
**os.environ,
|
||||||
|
"DOCKERHUB_REPO": dockerhub_repo,
|
||||||
|
"DOCKERHUB_IMG": dockerhub_img,
|
||||||
|
"RELEASE_VERSION": release_version,
|
||||||
|
"HUGGINGFACE_ACCESS_TOKEN": hf_token,
|
||||||
|
}
|
||||||
|
subprocess.run(
|
||||||
|
["docker", "buildx", "bake", "--push"],
|
||||||
|
check=True,
|
||||||
|
cwd=REPO_ROOT,
|
||||||
|
env=bake_env,
|
||||||
|
)
|
||||||
|
return tag
|
||||||
|
|
||||||
|
|
||||||
|
def api_request(method: str, url: str, api_key: str, body: dict | None = None) -> dict:
|
||||||
|
data = json.dumps(body).encode() if body is not None else None
|
||||||
|
req = urllib.request.Request(url, data=data, method=method)
|
||||||
|
req.add_header("Authorization", f"Bearer {api_key}")
|
||||||
|
req.add_header("Content-Type", "application/json")
|
||||||
|
try:
|
||||||
|
with urllib.request.urlopen(req) as resp:
|
||||||
|
raw = resp.read()
|
||||||
|
return json.loads(raw) if raw else {}
|
||||||
|
except urllib.error.HTTPError as e:
|
||||||
|
detail = e.read().decode(errors="replace")
|
||||||
|
raise RuntimeError(f"{method} {url} -> HTTP {e.code}: {detail}") from e
|
||||||
|
|
||||||
|
|
||||||
|
def create_template(api_key: str, name: str, image: str, env: dict, container_disk_in_gb: int) -> str:
|
||||||
|
log(f"Creating template {name!r} for image {image}")
|
||||||
|
resp = api_request("POST", f"{REST_API_BASE}/templates", api_key, {
|
||||||
|
"name": name,
|
||||||
|
"imageName": image,
|
||||||
|
"isServerless": True,
|
||||||
|
"env": env,
|
||||||
|
"containerDiskInGb": container_disk_in_gb,
|
||||||
|
})
|
||||||
|
return resp["id"]
|
||||||
|
|
||||||
|
|
||||||
|
def create_endpoint(
|
||||||
|
api_key: str,
|
||||||
|
name: str,
|
||||||
|
template_id: str,
|
||||||
|
gpu_type_ids: list[str],
|
||||||
|
gpu_count: int,
|
||||||
|
allowed_cuda_versions: list[str] | None,
|
||||||
|
min_cuda_version: str | None,
|
||||||
|
idle_timeout: int,
|
||||||
|
) -> str:
|
||||||
|
log(f"Creating endpoint {name!r} (gpuTypeIds={gpu_type_ids})")
|
||||||
|
body = {
|
||||||
|
"name": name,
|
||||||
|
"templateId": template_id,
|
||||||
|
"gpuTypeIds": gpu_type_ids,
|
||||||
|
"gpuCount": gpu_count,
|
||||||
|
"workersMin": 0,
|
||||||
|
"workersMax": 1,
|
||||||
|
"idleTimeout": idle_timeout,
|
||||||
|
"scalerType": "QUEUE_DELAY",
|
||||||
|
"scalerValue": 4,
|
||||||
|
}
|
||||||
|
if allowed_cuda_versions:
|
||||||
|
body["allowedCudaVersions"] = allowed_cuda_versions
|
||||||
|
if min_cuda_version:
|
||||||
|
body["minCudaVersion"] = min_cuda_version
|
||||||
|
resp = api_request("POST", f"{REST_API_BASE}/endpoints", api_key, body)
|
||||||
|
return resp["id"]
|
||||||
|
|
||||||
|
|
||||||
|
def delete_endpoint(api_key: str, endpoint_id: str) -> None:
|
||||||
|
log(f"Deleting endpoint {endpoint_id}")
|
||||||
|
try:
|
||||||
|
api_request("DELETE", f"{REST_API_BASE}/endpoints/{endpoint_id}", api_key)
|
||||||
|
except RuntimeError as e:
|
||||||
|
log(f"WARNING: failed to delete endpoint {endpoint_id}: {e}")
|
||||||
|
|
||||||
|
|
||||||
|
def delete_template(api_key: str, template_id: str) -> None:
|
||||||
|
log(f"Deleting template {template_id}")
|
||||||
|
try:
|
||||||
|
api_request("DELETE", f"{REST_API_BASE}/templates/{template_id}", api_key)
|
||||||
|
except RuntimeError as e:
|
||||||
|
log(f"WARNING: failed to delete template {template_id}: {e}")
|
||||||
|
|
||||||
|
|
||||||
|
def response_has_error(output) -> bool:
|
||||||
|
if isinstance(output, dict):
|
||||||
|
return "error" in output
|
||||||
|
if isinstance(output, list):
|
||||||
|
return any(isinstance(item, dict) and "error" in item for item in output)
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def run_test_case(api_key: str, endpoint_id: str, test: dict, cold_start_buffer_seconds: int) -> bool:
|
||||||
|
name = test.get("name", "unnamed_test")
|
||||||
|
deadline = time.monotonic() + test.get("timeout", 300000) / 1000 + cold_start_buffer_seconds
|
||||||
|
|
||||||
|
log(f"Submitting job for test {name!r}")
|
||||||
|
submit = api_request("POST", f"{JOB_API_BASE}/{endpoint_id}/run", api_key, {"input": test["input"]})
|
||||||
|
job_id = submit["id"]
|
||||||
|
|
||||||
|
while True:
|
||||||
|
if time.monotonic() > deadline:
|
||||||
|
log(f"FAIL {name}: timed out waiting for job {job_id}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
status_resp = api_request("GET", f"{JOB_API_BASE}/{endpoint_id}/status/{job_id}", api_key)
|
||||||
|
status = status_resp.get("status")
|
||||||
|
|
||||||
|
if status in TERMINAL_STATUSES:
|
||||||
|
if status != "COMPLETED":
|
||||||
|
log(f"FAIL {name}: job {job_id} ended with status {status}: {status_resp}")
|
||||||
|
return False
|
||||||
|
if response_has_error(status_resp.get("output")):
|
||||||
|
log(f"FAIL {name}: job {job_id} completed but output contained an error: {status_resp.get('output')}")
|
||||||
|
return False
|
||||||
|
log(f"PASS {name}")
|
||||||
|
return True
|
||||||
|
|
||||||
|
time.sleep(5)
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument("--config", required=True, type=Path, help="Path to a configs/**/*.yaml vLLM config")
|
||||||
|
parser.add_argument("--image", help="Existing image tag to test; skips building unless --build is also given")
|
||||||
|
parser.add_argument("--build", action="store_true", help="Build and push the image before testing")
|
||||||
|
parser.add_argument("--keep", action="store_true", help="Don't tear down the endpoint/template afterward")
|
||||||
|
parser.add_argument("--hub-json", type=Path, default=REPO_ROOT / ".runpod" / "hub.json")
|
||||||
|
parser.add_argument("--tests-file", type=Path, default=REPO_ROOT / ".runpod" / "tests.json")
|
||||||
|
parser.add_argument("--gpu-type-ids", default=",".join(DEFAULT_GPU_TYPE_IDS))
|
||||||
|
parser.add_argument("--min-cuda-version", help="Overrides the config/hub.json minCudaVersion")
|
||||||
|
parser.add_argument("--dockerhub-repo", default=os.environ.get("DOCKERHUB_REPO", "runpod"))
|
||||||
|
parser.add_argument("--dockerhub-img", default=os.environ.get("DOCKERHUB_IMG", "worker-v1-vllm"))
|
||||||
|
parser.add_argument("--idle-timeout", type=int, default=60)
|
||||||
|
parser.add_argument("--cold-start-buffer-seconds", type=int, default=600)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
api_key = os.environ.get("RUNPOD_API_KEY")
|
||||||
|
if not api_key:
|
||||||
|
log("ERROR: RUNPOD_API_KEY is not set")
|
||||||
|
return 1
|
||||||
|
|
||||||
|
model_slug = args.config.stem
|
||||||
|
run_id = uuid.uuid4().hex[:8]
|
||||||
|
|
||||||
|
if args.build or not args.image:
|
||||||
|
release_version = f"test-{model_slug}-{run_id}"
|
||||||
|
image = build_and_push_image(args.dockerhub_repo, args.dockerhub_img, release_version)
|
||||||
|
else:
|
||||||
|
image = args.image
|
||||||
|
|
||||||
|
hub_defaults = load_hub_defaults(args.hub_json)
|
||||||
|
model_env = load_model_env(args.config)
|
||||||
|
env = {**hub_defaults["env"], **model_env}
|
||||||
|
hf_token = os.environ.get("HF_TOKEN") or os.environ.get("HUGGINGFACE_ACCESS_TOKEN")
|
||||||
|
if hf_token:
|
||||||
|
env["HF_TOKEN"] = hf_token
|
||||||
|
tests_data = json.loads(args.tests_file.read_text())
|
||||||
|
gpu_type_ids = [g.strip() for g in args.gpu_type_ids.split(",") if g.strip()]
|
||||||
|
|
||||||
|
resource_name = f"worker-vllm-e2e-{model_slug}-{run_id}"
|
||||||
|
template_id = None
|
||||||
|
endpoint_id = None
|
||||||
|
try:
|
||||||
|
template_id = create_template(
|
||||||
|
api_key, resource_name, image, env, hub_defaults["containerDiskInGb"]
|
||||||
|
)
|
||||||
|
write_github_output("template_id", template_id)
|
||||||
|
endpoint_id = create_endpoint(
|
||||||
|
api_key,
|
||||||
|
resource_name,
|
||||||
|
template_id,
|
||||||
|
gpu_type_ids,
|
||||||
|
hub_defaults["gpuCount"],
|
||||||
|
hub_defaults["allowedCudaVersions"],
|
||||||
|
args.min_cuda_version or hub_defaults["minCudaVersion"],
|
||||||
|
args.idle_timeout,
|
||||||
|
)
|
||||||
|
write_github_output("endpoint_id", endpoint_id)
|
||||||
|
|
||||||
|
results = [
|
||||||
|
run_test_case(api_key, endpoint_id, test, args.cold_start_buffer_seconds)
|
||||||
|
for test in tests_data["tests"]
|
||||||
|
]
|
||||||
|
|
||||||
|
if all(results):
|
||||||
|
log(f"All {len(results)} test(s) passed for {model_slug}")
|
||||||
|
return 0
|
||||||
|
log(f"{results.count(False)}/{len(results)} test(s) failed for {model_slug}")
|
||||||
|
return 1
|
||||||
|
finally:
|
||||||
|
if not args.keep:
|
||||||
|
if endpoint_id:
|
||||||
|
delete_endpoint(api_key, endpoint_id)
|
||||||
|
if template_id:
|
||||||
|
delete_template(api_key, template_id)
|
||||||
|
else:
|
||||||
|
log(f"--keep passed; leaving endpoint={endpoint_id} template={template_id} running")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
@@ -9,7 +9,6 @@ from typing import AsyncGenerator, Optional
|
|||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
from vllm import AsyncLLMEngine
|
from vllm import AsyncLLMEngine
|
||||||
from vllm.inputs import TextPrompt
|
from vllm.inputs import TextPrompt
|
||||||
from vllm.entrypoints.logger import RequestLogger
|
|
||||||
from vllm.entrypoints.anthropic.protocol import AnthropicMessagesRequest, AnthropicMessagesResponse, AnthropicError, AnthropicErrorResponse
|
from vllm.entrypoints.anthropic.protocol import AnthropicMessagesRequest, AnthropicMessagesResponse, AnthropicError, AnthropicErrorResponse
|
||||||
from vllm.entrypoints.anthropic.serving import AnthropicServingMessages
|
from vllm.entrypoints.anthropic.serving import AnthropicServingMessages
|
||||||
from vllm.entrypoints.openai.chat_completion.protocol import ChatCompletionRequest
|
from vllm.entrypoints.openai.chat_completion.protocol import ChatCompletionRequest
|
||||||
|
|||||||
+28
-1
@@ -404,6 +404,23 @@ def _resolve_cached_model_path(model_name: str) -> str:
|
|||||||
return resolved
|
return resolved
|
||||||
|
|
||||||
|
|
||||||
|
def _get_args_from_config_file() -> dict:
|
||||||
|
"""Load engine args from a vLLM-style config.yaml.
|
||||||
|
|
||||||
|
Checks VLLM_CONFIG_FILE env var, then falls back to /vllm_config.yaml.
|
||||||
|
Keys use the same long-form names as vllm serve (hyphens converted to underscores).
|
||||||
|
"""
|
||||||
|
import yaml
|
||||||
|
path = os.getenv("VLLM_CONFIG_FILE", "/vllm_config.yaml")
|
||||||
|
if not os.path.exists(path):
|
||||||
|
return {}
|
||||||
|
with open(path) as f:
|
||||||
|
raw = yaml.safe_load(f) or {}
|
||||||
|
normalized = {k.replace("-", "_"): v for k, v in raw.items()}
|
||||||
|
logging.info("Loaded engine args from config file %s: %s", path, list(normalized.keys()))
|
||||||
|
return normalized
|
||||||
|
|
||||||
|
|
||||||
def get_local_args():
|
def get_local_args():
|
||||||
"""
|
"""
|
||||||
Retrieve local arguments from a JSON file.
|
Retrieve local arguments from a JSON file.
|
||||||
@@ -429,6 +446,9 @@ def get_engine_args():
|
|||||||
# Start with worker custom defaults (only where we differ from vLLM)
|
# Start with worker custom defaults (only where we differ from vLLM)
|
||||||
args = dict(DEFAULT_ARGS)
|
args = dict(DEFAULT_ARGS)
|
||||||
|
|
||||||
|
# Config file values sit above defaults but below env vars
|
||||||
|
args.update(_get_args_from_config_file())
|
||||||
|
|
||||||
# Auto-discover: every AsyncEngineArgs field from env UPPERCASED (e.g. MAX_MODEL_LEN)
|
# Auto-discover: every AsyncEngineArgs field from env UPPERCASED (e.g. MAX_MODEL_LEN)
|
||||||
args.update(_get_args_from_env_auto_discover())
|
args.update(_get_args_from_env_auto_discover())
|
||||||
|
|
||||||
@@ -581,6 +601,13 @@ def get_engine_args():
|
|||||||
|
|
||||||
# Resolve lowercase HF cache paths (FDE-174)
|
# Resolve lowercase HF cache paths (FDE-174)
|
||||||
if args.get("model"):
|
if args.get("model"):
|
||||||
args["model"] = _resolve_cached_model_path(args["model"])
|
original_model = args["model"]
|
||||||
|
args["model"] = _resolve_cached_model_path(original_model)
|
||||||
|
# When the model was rewritten to an on-disk snapshot path, keep serving
|
||||||
|
# under the original repo id so the OpenAI API model name does not become
|
||||||
|
# a filesystem path (issue #310). An explicit served_model_name (or the
|
||||||
|
# OPENAI_SERVED_MODEL_NAME_OVERRIDE handled downstream) still wins.
|
||||||
|
if args["model"] != original_model and not args.get("served_model_name"):
|
||||||
|
args["served_model_name"] = original_model
|
||||||
|
|
||||||
return AsyncEngineArgs(**args)
|
return AsyncEngineArgs(**args)
|
||||||
|
|||||||
@@ -0,0 +1,105 @@
|
|||||||
|
"""Shared test fixtures.
|
||||||
|
|
||||||
|
``src/engine_args.py`` hard-imports ``vllm`` (and a tensorizer submodule) and
|
||||||
|
``torch.cuda``. Both are only installed inside the GPU Docker image, so when the
|
||||||
|
tests run on a machine without them we install lightweight stubs. When the real
|
||||||
|
packages *are* available (e.g. CI inside the worker image) the stubs are skipped
|
||||||
|
and the real ones are used instead.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import sys
|
||||||
|
import types
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from typing import Optional, Union, List
|
||||||
|
|
||||||
|
|
||||||
|
def _install_torch_stub():
|
||||||
|
try:
|
||||||
|
import torch # noqa: F401
|
||||||
|
return # real torch present, nothing to stub
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
torch = types.ModuleType("torch")
|
||||||
|
cuda = types.ModuleType("torch.cuda")
|
||||||
|
# No GPU in the test environment -> 0 devices (skips tensor-parallel setup).
|
||||||
|
cuda.device_count = lambda: 0
|
||||||
|
torch.cuda = cuda
|
||||||
|
sys.modules["torch"] = torch
|
||||||
|
sys.modules["torch.cuda"] = cuda
|
||||||
|
|
||||||
|
|
||||||
|
def _install_vllm_stub():
|
||||||
|
try:
|
||||||
|
import vllm # noqa: F401
|
||||||
|
return # real vLLM present, nothing to stub
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
vllm = types.ModuleType("vllm")
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class AsyncEngineArgs:
|
||||||
|
# Only the fields the worker actually sets/reads need to exist here;
|
||||||
|
# get_engine_args() filters args down to AsyncEngineArgs.__dataclass_fields__
|
||||||
|
# before construction, so unknown keys are dropped rather than passed.
|
||||||
|
model: Optional[str] = None
|
||||||
|
served_model_name: Optional[Union[str, List[str]]] = None
|
||||||
|
revision: Optional[str] = None
|
||||||
|
tokenizer: Optional[str] = None
|
||||||
|
trust_remote_code: bool = False
|
||||||
|
max_model_len: Optional[int] = None
|
||||||
|
max_num_batched_tokens: Optional[int] = None
|
||||||
|
disable_log_stats: bool = False
|
||||||
|
gpu_memory_utilization: float = 0.9
|
||||||
|
tensor_parallel_size: int = 1
|
||||||
|
max_parallel_loading_workers: Optional[int] = None
|
||||||
|
kv_cache_dtype: Optional[str] = None
|
||||||
|
|
||||||
|
class _Stub: # pragma: no cover - placeholder for vllm symbols
|
||||||
|
def __init__(self, *args, **kwargs):
|
||||||
|
pass
|
||||||
|
|
||||||
|
vllm.AsyncEngineArgs = AsyncEngineArgs
|
||||||
|
vllm.SamplingParams = _Stub
|
||||||
|
sys.modules["vllm"] = vllm
|
||||||
|
|
||||||
|
# src.utils imports these at module load and uses ErrorResponse as a return
|
||||||
|
# annotation, which Python evaluates eagerly on <3.14 -> must be defined.
|
||||||
|
vllm_utils = types.ModuleType("vllm.utils")
|
||||||
|
vllm_utils.random_uuid = lambda: "stub-uuid"
|
||||||
|
vllm.utils = vllm_utils
|
||||||
|
sys.modules["vllm.utils"] = vllm_utils
|
||||||
|
|
||||||
|
protocol = types.ModuleType("vllm.entrypoints.openai.engine.protocol")
|
||||||
|
protocol.ErrorResponse = _Stub
|
||||||
|
protocol.ErrorInfo = _Stub
|
||||||
|
protocol.RequestResponseMetadata = _Stub
|
||||||
|
for name in (
|
||||||
|
"vllm.entrypoints",
|
||||||
|
"vllm.entrypoints.openai",
|
||||||
|
"vllm.entrypoints.openai.engine",
|
||||||
|
):
|
||||||
|
sys.modules.setdefault(name, types.ModuleType(name))
|
||||||
|
sys.modules["vllm.entrypoints.openai.engine.protocol"] = protocol
|
||||||
|
|
||||||
|
# vllm.model_executor.model_loader.tensorizer.TensorizerConfig
|
||||||
|
model_executor = types.ModuleType("vllm.model_executor")
|
||||||
|
model_loader = types.ModuleType("vllm.model_executor.model_loader")
|
||||||
|
tensorizer = types.ModuleType("vllm.model_executor.model_loader.tensorizer")
|
||||||
|
|
||||||
|
class TensorizerConfig: # pragma: no cover - placeholder
|
||||||
|
def __init__(self, *args, **kwargs):
|
||||||
|
pass
|
||||||
|
|
||||||
|
tensorizer.TensorizerConfig = TensorizerConfig
|
||||||
|
model_loader.tensorizer = tensorizer
|
||||||
|
model_executor.model_loader = model_loader
|
||||||
|
vllm.model_executor = model_executor
|
||||||
|
sys.modules["vllm.model_executor"] = model_executor
|
||||||
|
sys.modules["vllm.model_executor.model_loader"] = model_loader
|
||||||
|
sys.modules["vllm.model_executor.model_loader.tensorizer"] = tensorizer
|
||||||
|
|
||||||
|
|
||||||
|
_install_torch_stub()
|
||||||
|
_install_vllm_stub()
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
# Test-only dependencies. vllm/torch are stubbed in conftest.py when absent,
|
||||||
|
# so the unit tests run on a plain CPU runner without the GPU image.
|
||||||
|
pytest>=8,<10
|
||||||
|
# get_engine_args() reads a vLLM-style config via PyYAML (a transitive vllm dep
|
||||||
|
# at runtime); install it explicitly here since vllm itself is stubbed.
|
||||||
|
pyyaml
|
||||||
@@ -0,0 +1,126 @@
|
|||||||
|
"""Tests for HF cache path resolution and served-model-name decoupling.
|
||||||
|
|
||||||
|
Regression coverage for issue #310: when MODEL_NAME is served from a lowercased
|
||||||
|
HF cache dir, the cache resolver rewrites engine_args.model to a snapshot path.
|
||||||
|
The served model name must stay the original repo id, not the path.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from src import engine_args
|
||||||
|
from src.engine_args import _resolve_cached_model_path, get_engine_args
|
||||||
|
|
||||||
|
|
||||||
|
MODEL = "Qwen/Qwen3.6-27B-FP8"
|
||||||
|
SNAPSHOT_HASH = "e89b16ebf1988b3d6befa7de50abc2d76f26eb09"
|
||||||
|
|
||||||
|
|
||||||
|
def _make_cache(root, folder_name, snapshot=SNAPSHOT_HASH):
|
||||||
|
"""Create a HF-style ``models--…/snapshots/<hash>/`` dir and return its path."""
|
||||||
|
snap_dir = os.path.join(root, folder_name, "snapshots", snapshot)
|
||||||
|
os.makedirs(snap_dir)
|
||||||
|
return snap_dir
|
||||||
|
|
||||||
|
|
||||||
|
def _is_case_sensitive_fs(path):
|
||||||
|
"""The lowercase-cache resolution only matters on case-sensitive filesystems.
|
||||||
|
|
||||||
|
On macOS (APFS, case-insensitive by default) ``models--Qwen--…`` and
|
||||||
|
``models--qwen--…`` collide, so the resolver always sees the exact-case dir
|
||||||
|
as present. Production runs on Linux (case-sensitive), which is what these
|
||||||
|
tests exercise.
|
||||||
|
"""
|
||||||
|
probe = os.path.join(path, "CaseProbe")
|
||||||
|
open(probe, "w").close()
|
||||||
|
try:
|
||||||
|
return not os.path.exists(os.path.join(path, "caseprobe"))
|
||||||
|
finally:
|
||||||
|
os.remove(probe)
|
||||||
|
|
||||||
|
|
||||||
|
requires_case_sensitive_fs = pytest.mark.skipif(
|
||||||
|
not _is_case_sensitive_fs(os.environ.get("TMPDIR", "/tmp")),
|
||||||
|
reason="lowercase HF cache resolution only applies on case-sensitive filesystems",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def hf_cache(tmp_path, monkeypatch):
|
||||||
|
cache = tmp_path / "hub"
|
||||||
|
cache.mkdir()
|
||||||
|
monkeypatch.setenv("HUGGINGFACE_HUB_CACHE", str(cache))
|
||||||
|
# Make sure HF_HOME does not shadow the explicit cache dir during the test.
|
||||||
|
monkeypatch.delenv("HF_HOME", raising=False)
|
||||||
|
return cache
|
||||||
|
|
||||||
|
|
||||||
|
class TestResolveCachedModelPath:
|
||||||
|
def test_exact_case_dir_returns_repo_id(self, hf_cache):
|
||||||
|
_make_cache(str(hf_cache), "models--Qwen--Qwen3.6-27B-FP8")
|
||||||
|
assert _resolve_cached_model_path(MODEL) == MODEL
|
||||||
|
|
||||||
|
def test_no_cache_returns_repo_id(self, hf_cache):
|
||||||
|
assert _resolve_cached_model_path(MODEL) == MODEL
|
||||||
|
|
||||||
|
def test_absolute_path_passthrough(self, hf_cache):
|
||||||
|
path = "/runpod-volume/some/local/model"
|
||||||
|
assert _resolve_cached_model_path(path) == path
|
||||||
|
|
||||||
|
@requires_case_sensitive_fs
|
||||||
|
def test_lowercase_dir_returns_snapshot_path(self, hf_cache):
|
||||||
|
snap = _make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")
|
||||||
|
assert _resolve_cached_model_path(MODEL) == snap
|
||||||
|
|
||||||
|
def test_lowercase_dir_without_snapshots_returns_repo_id(self, hf_cache):
|
||||||
|
# Dir exists but has no snapshots subdir -> nothing to resolve to.
|
||||||
|
os.makedirs(os.path.join(str(hf_cache), "models--qwen--qwen3.6-27b-fp8"))
|
||||||
|
assert _resolve_cached_model_path(MODEL) == MODEL
|
||||||
|
|
||||||
|
@requires_case_sensitive_fs
|
||||||
|
def test_lowercase_dir_picks_latest_snapshot(self, hf_cache):
|
||||||
|
folder = "models--qwen--qwen3.6-27b-fp8"
|
||||||
|
_make_cache(str(hf_cache), folder, snapshot="aaaa")
|
||||||
|
latest = _make_cache(str(hf_cache), folder, snapshot="zzzz")
|
||||||
|
assert _resolve_cached_model_path(MODEL) == latest
|
||||||
|
|
||||||
|
|
||||||
|
class TestGetEngineArgsServedName:
|
||||||
|
"""Issue #310: served name must be decoupled from the resolved on-disk path."""
|
||||||
|
|
||||||
|
@pytest.fixture(autouse=True)
|
||||||
|
def base_env(self, monkeypatch):
|
||||||
|
# Avoid the network branch in _resolve_max_model_len.
|
||||||
|
monkeypatch.setenv("MAX_NUM_BATCHED_TOKENS", "2048")
|
||||||
|
monkeypatch.delenv("SERVED_MODEL_NAME", raising=False)
|
||||||
|
# Don't pick up a stray vLLM config file from the environment.
|
||||||
|
monkeypatch.setenv("VLLM_CONFIG_FILE", "/nonexistent-vllm-config.yaml")
|
||||||
|
|
||||||
|
@requires_case_sensitive_fs
|
||||||
|
def test_served_name_is_repo_id_when_path_rewritten(self, hf_cache, monkeypatch):
|
||||||
|
snap = _make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")
|
||||||
|
monkeypatch.setenv("MODEL_NAME", MODEL)
|
||||||
|
|
||||||
|
result = get_engine_args()
|
||||||
|
|
||||||
|
assert result.model == snap # weights load from the lowercase cache
|
||||||
|
assert result.served_model_name == MODEL # API still serves the repo id
|
||||||
|
|
||||||
|
def test_served_name_untouched_when_no_rewrite(self, hf_cache, monkeypatch):
|
||||||
|
_make_cache(str(hf_cache), "models--Qwen--Qwen3.6-27B-FP8")
|
||||||
|
monkeypatch.setenv("MODEL_NAME", MODEL)
|
||||||
|
|
||||||
|
result = get_engine_args()
|
||||||
|
|
||||||
|
assert result.model == MODEL
|
||||||
|
assert result.served_model_name is None
|
||||||
|
|
||||||
|
def test_explicit_served_name_not_overridden(self, hf_cache, monkeypatch):
|
||||||
|
_make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")
|
||||||
|
monkeypatch.setenv("MODEL_NAME", MODEL)
|
||||||
|
monkeypatch.setenv("SERVED_MODEL_NAME", "custom-name")
|
||||||
|
|
||||||
|
result = get_engine_args()
|
||||||
|
|
||||||
|
assert result.served_model_name == "custom-name"
|
||||||
Reference in New Issue
Block a user