Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
76f345be08 | ||
|
|
4dfda80fd0 | ||
|
|
89f25b0d70 | ||
|
|
4808160735 | ||
|
|
51e1aee770 | ||
|
|
86b7951c39 | ||
|
|
e786204cf5 | ||
|
|
591f9d6531 | ||
|
|
49967d0c09 | ||
|
|
e97cab854f | ||
|
|
828e797498 | ||
|
|
373847d9f9 | ||
|
|
2000acc0a3 | ||
|
|
89e3b2e920 | ||
|
|
85f8c81c1c | ||
|
|
88490e9551 | ||
|
|
798f2b12eb | ||
|
|
9e1c483136 | ||
|
|
d71ea9939d | ||
|
|
7e2b4e2288 | ||
|
|
015f8f3c4d | ||
|
|
75ffcf73f2 | ||
|
|
d7ba3b6ab7 | ||
|
|
b11c91722c | ||
|
|
fcdc799e0d | ||
|
|
84ec446493 | ||
|
|
db246653a2 | ||
|
|
1b3228a2dc | ||
|
|
4817d4a8e7 | ||
|
|
0378382a92 | ||
|
|
08580e7ccf | ||
|
|
8868aae6b1 | ||
|
|
fb8adc5c06 | ||
|
|
7351da512b | ||
|
|
1e78043b2c | ||
|
|
352c64f4c1 | ||
|
|
0922f5b435 | ||
|
|
5d9a48fc70 | ||
|
|
5d1579e361 | ||
|
|
0488b77d89 | ||
|
|
d3a962c33b | ||
|
|
105c125698 | ||
|
|
0a0ccfcb60 | ||
|
|
c8ce53c72c | ||
|
|
9618e799ba | ||
|
|
cb3f077dba | ||
|
|
69646b9e99 | ||
|
|
8b991a7ad7 | ||
|
|
dac05b62b3 | ||
|
|
d356c31675 | ||
|
|
14b74a4989 | ||
|
|
80072047ab | ||
|
|
50aba8fb57 | ||
|
|
9edc5715ce | ||
|
|
4c91f2c5b5 | ||
|
|
6265b99348 | ||
|
|
026f8d700b | ||
|
|
146bdb0252 | ||
|
|
da01193a3d | ||
|
|
c2e6cc9f61 | ||
|
|
69968a6b39 | ||
|
|
32b29d4c6c | ||
|
|
dcea4fc4f9 | ||
|
|
9c139e8ceb | ||
|
|
678bb4be8f | ||
|
|
87d7365126 | ||
|
|
0e83616f93 | ||
|
|
ab6d39dcf8 | ||
|
|
ed315a175e | ||
|
|
73f030ae5e | ||
|
|
8a099c1723 | ||
|
|
ff87840a58 | ||
|
|
7dc853b1fe | ||
|
|
a8b754b92a | ||
|
|
6357aeda51 | ||
|
|
0140b29c44 | ||
|
|
0cb8aeae77 | ||
|
|
cd8f9e9560 | ||
|
|
895fd25fac | ||
|
|
7bb8df73af | ||
|
|
747cdf5891 | ||
|
|
3d4af5df9b | ||
|
|
72547aa3bb | ||
|
|
577fd8c3c3 | ||
|
|
f49f35456e | ||
|
|
cff7b09ef2 | ||
|
|
04b342c675 | ||
|
|
5b29643799 | ||
|
|
4f8a16df5d | ||
|
|
22356ee2b3 | ||
|
|
fa42ecd79a | ||
|
|
178c72238e | ||
|
|
e6950bdebd | ||
|
|
a774cefe85 | ||
|
|
9de17d49b7 | ||
|
|
296556a6f7 | ||
|
|
a1544ea70d | ||
|
|
3403889528 | ||
|
|
f299204770 | ||
|
|
1ed25eea20 | ||
|
|
dc4ad7ddeb | ||
|
|
9035b0e07f | ||
|
|
c979f0020f | ||
|
|
6fbd480a26 | ||
|
|
3ef1fb8e7b | ||
|
|
30e8514d63 | ||
|
|
d9808815ee | ||
|
|
d8ed3b5353 | ||
|
|
4c4e039565 | ||
|
|
9d1686960d | ||
|
|
45d1eeee47 | ||
|
|
17efb0e7d0 | ||
|
|
2b5f07df63 | ||
|
|
13fa71878e | ||
|
|
8a9365bed4 | ||
|
|
cd485a1af1 | ||
|
|
b9043639e9 | ||
|
|
407dbd7773 | ||
|
|
f103c142c1 | ||
|
|
efb093e198 | ||
|
|
42443f735e | ||
|
|
b7c6d4f9a2 | ||
|
|
d69cc021e8 | ||
|
|
61faa8f137 | ||
|
|
1606cff557 | ||
|
|
e705c9494b | ||
|
|
b749aa5718 | ||
|
|
4705ba8a7c | ||
|
|
767c66c301 | ||
|
|
fefdbe21a9 | ||
|
|
ee961ad28d | ||
|
|
2e8c251447 | ||
|
|
c3cf43b228 | ||
|
|
7ec10b98cd | ||
|
|
c45ac42acd | ||
|
|
340bc0b3c6 | ||
|
|
e1e9ef74ad | ||
|
|
461f89cea6 | ||
|
|
8eb55b90c1 | ||
|
|
6d6cbe7095 | ||
|
|
90c16b472d | ||
|
|
6f2381a9a1 |
@@ -0,0 +1,71 @@
|
||||
name: CI | Sync vLLM version in READMEs
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: ["main"]
|
||||
paths:
|
||||
- "Dockerfile"
|
||||
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
|
||||
jobs:
|
||||
sync_version:
|
||||
runs-on: ubuntu-latest
|
||||
name: Check README version matches Dockerfile and update if needed
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Extract vLLM version from Dockerfile and sync READMEs
|
||||
run: |
|
||||
echo "Extracting vLLM version from Dockerfile..."
|
||||
dockerfile_version=$(grep -oP 'vllm(?:\[[\w,]+\])?==\K[\d.]+' Dockerfile | head -1)
|
||||
|
||||
if [ -z "$dockerfile_version" ]; then
|
||||
echo "ERROR: Could not extract vLLM version from Dockerfile."
|
||||
exit 1
|
||||
fi
|
||||
echo "Dockerfile vLLM version: $dockerfile_version"
|
||||
echo "VLLM_VERSION=$dockerfile_version" >> $GITHUB_ENV
|
||||
|
||||
updated=0
|
||||
for readme in README.md .runpod/README.md; do
|
||||
if [ ! -f "$readme" ]; then
|
||||
echo "Skipping $readme (not found)"
|
||||
continue
|
||||
fi
|
||||
|
||||
readme_version=$(grep -oP 'Current vLLM version: \[\K[\d.]+' "$readme" || echo "")
|
||||
echo "$readme current version: ${readme_version:-not found}"
|
||||
|
||||
if [ "$readme_version" = "$dockerfile_version" ]; then
|
||||
echo "$readme is already up to date."
|
||||
continue
|
||||
fi
|
||||
|
||||
echo "Updating $readme from $readme_version to $dockerfile_version..."
|
||||
sed -i "s|Current vLLM version: \[${readme_version}\](https://github.com/vllm-project/vllm/releases/tag/v${readme_version})|Current vLLM version: [${dockerfile_version}](https://github.com/vllm-project/vllm/releases/tag/v${dockerfile_version})|g" "$readme"
|
||||
updated=1
|
||||
done
|
||||
|
||||
echo "UPDATED=$updated" >> $GITHUB_ENV
|
||||
|
||||
- name: Create Pull Request
|
||||
if: env.UPDATED == '1'
|
||||
uses: peter-evans/create-pull-request@v7
|
||||
with:
|
||||
token: ${{ secrets.GITHUB_TOKEN }}
|
||||
commit-message: "docs: sync vLLM version to ${{ env.VLLM_VERSION }} in READMEs"
|
||||
title: "docs: sync vLLM version to ${{ env.VLLM_VERSION }} in READMEs"
|
||||
body: |
|
||||
The vLLM version in the Dockerfile has been updated to `${{ env.VLLM_VERSION }}`.
|
||||
|
||||
This PR syncs the version badge/link in:
|
||||
- `README.md`
|
||||
- `.runpod/README.md`
|
||||
branch: docs/sync-vllm-version-${{ env.VLLM_VERSION }}
|
||||
labels: documentation
|
||||
@@ -9,59 +9,60 @@ on:
|
||||
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
|
||||
jobs:
|
||||
check_dep:
|
||||
runs-on: ubuntu-latest
|
||||
name: Check python requirements file and update
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v2
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Check for new package version and update
|
||||
run: |
|
||||
echo "Fetching the current runpod version from requirements.txt..."
|
||||
echo "Fetching current runpod version from requirements.txt..."
|
||||
|
||||
# Get current version, allowing both == and ~= in the search pattern
|
||||
current_version=$(grep -oP 'runpod[~=]{1,2}\K[^"]+' ./builder/requirements.txt)
|
||||
echo "Current version: $current_version"
|
||||
# Match runpod with any version specifier or no specifier at all
|
||||
current_version=$(grep -oP '^runpod([~>=!<]{1,2}\K[\d.]+)?' ./builder/requirements.txt | grep -oP '[\d.]+' || echo "")
|
||||
echo "Current version: ${current_version:-unset}"
|
||||
|
||||
# Extract major and minor from current version
|
||||
current_major_minor=$(echo $current_version | cut -d. -f1,2)
|
||||
echo "Current major.minor: $current_major_minor"
|
||||
|
||||
echo "Fetching the latest runpod version from PyPI..."
|
||||
|
||||
# Get new version from PyPI
|
||||
new_version=$(curl -s https://pypi.org/pypi/runpod/json | jq -r .info.version)
|
||||
echo "Fetching latest runpod version from PyPI..."
|
||||
new_version=$(curl -sf https://pypi.org/pypi/runpod/json | jq -r .info.version)
|
||||
echo "NEW_VERSION_ENV=$new_version" >> $GITHUB_ENV
|
||||
echo "New version: $new_version"
|
||||
|
||||
# Extract major and minor from new version
|
||||
new_major_minor=$(echo $new_version | cut -d. -f1,2)
|
||||
echo "New major.minor: $new_major_minor"
|
||||
|
||||
if [ -z "$new_version" ]; then
|
||||
echo "ERROR: Failed to fetch the new version from PyPI."
|
||||
exit 1
|
||||
echo "ERROR: Failed to fetch new version from PyPI."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Check if the major or minor version is different
|
||||
if [ "$current_major_minor" = "$new_major_minor" ]; then
|
||||
echo "No update needed. The new version ($new_major_minor) is within the allowed range (~= $current_major_minor)."
|
||||
if [ -z "$current_version" ]; then
|
||||
echo "No version pin found — pinning to $new_version."
|
||||
else
|
||||
current_major_minor=$(echo "$current_version" | cut -d. -f1,2)
|
||||
new_major_minor=$(echo "$new_version" | cut -d. -f1,2)
|
||||
echo "Current major.minor: $current_major_minor New major.minor: $new_major_minor"
|
||||
|
||||
if [ "$current_major_minor" = "$new_major_minor" ]; then
|
||||
echo "No update needed. New version ($new_version) is within ~= $current_major_minor range."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "New major/minor detected ($new_major_minor). Updating requirements.txt..."
|
||||
fi
|
||||
|
||||
echo "New major/minor detected ($new_major_minor). Updating requirements.txt..."
|
||||
|
||||
# Update requirements.txt, preserving the existing constraint type (~= or ==)
|
||||
sed -i "s/runpod[~=][^ ]*/runpod~=$new_version/" ./builder/requirements.txt
|
||||
echo "requirements.txt has been updated."
|
||||
# Replace any `runpod`, `runpod==x`, `runpod~=x`, etc. with pinned version
|
||||
sed -i "s|^runpod.*|runpod~=$new_version|" ./builder/requirements.txt
|
||||
echo "requirements.txt updated."
|
||||
|
||||
- name: Create Pull Request
|
||||
uses: peter-evans/create-pull-request@v3
|
||||
uses: peter-evans/create-pull-request@v7
|
||||
with:
|
||||
token: ${{ secrets.GITHUB_TOKEN }}
|
||||
commit-message: Update runpod package version
|
||||
title: Update runpod package version
|
||||
body: The package version has been updated to ${{ env.NEW_VERSION_ENV }}
|
||||
commit-message: "chore: update runpod to ${{ env.NEW_VERSION_ENV }}"
|
||||
title: "chore: update runpod to ${{ env.NEW_VERSION_ENV }}"
|
||||
body: The `runpod` package has been updated to `${{ env.NEW_VERSION_ENV }}`.
|
||||
branch: runpod-package-update
|
||||
|
||||
@@ -3,7 +3,7 @@ name: Release
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- "v[0-9]+.[0-9]+.[0-9]+*" # Trigger on version tags like v1.0.0, v2.1.0, etc.
|
||||
- "v[0-9]+.[0-9]+.[0-9]+*"
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
version:
|
||||
@@ -53,16 +53,13 @@ jobs:
|
||||
|
||||
# Determine version based on trigger type
|
||||
if [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then
|
||||
# Manual trigger: use input version
|
||||
VERSION="${{ github.event.inputs.version }}"
|
||||
echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV
|
||||
echo "IS_MANUAL_RELEASE=true" >> $GITHUB_ENV
|
||||
elif [[ "${{ github.event_name }}" == "release" ]]; then
|
||||
VERSION="${{ github.event.release.tag_name }}"
|
||||
else
|
||||
# Tag trigger: use tag name (remove refs/tags/ prefix)
|
||||
VERSION=${GITHUB_REF#refs/tags/}
|
||||
echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV
|
||||
echo "IS_MANUAL_RELEASE=false" >> $GITHUB_ENV
|
||||
fi
|
||||
echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV
|
||||
|
||||
- name: Build and push the images to Docker Hub
|
||||
uses: docker/bake-action@v2
|
||||
@@ -76,11 +73,45 @@ jobs:
|
||||
|
||||
- name: Release Summary
|
||||
run: |
|
||||
echo "🚀 Release completed!"
|
||||
echo "Release completed!"
|
||||
echo "Version: ${{ env.RELEASE_VERSION }}"
|
||||
echo "Docker Image: ${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}"
|
||||
if [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then
|
||||
echo "Trigger: Manual workflow dispatch"
|
||||
else
|
||||
echo "Trigger: GitHub release (tag: ${{ github.ref_name }})"
|
||||
|
||||
- name: Fetch Release Notes
|
||||
run: |
|
||||
RESPONSE=$(curl -sf \
|
||||
-H "Authorization: token ${{ github.token }}" \
|
||||
"https://api.github.com/repos/${{ github.repository }}/releases/tags/${{ env.RELEASE_VERSION }}" 2>/dev/null) || true
|
||||
if [[ -n "$RESPONSE" ]]; then
|
||||
NOTES=$(echo "$RESPONSE" | jq -r '.body // empty')
|
||||
fi
|
||||
printf '%s' "${NOTES:-No release notes available.}" > /tmp/release_notes.txt
|
||||
|
||||
- name: Notify Slack
|
||||
run: |
|
||||
jq -n \
|
||||
--arg version "${{ env.RELEASE_VERSION }}" \
|
||||
--arg docker "${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}" \
|
||||
--rawfile notes /tmp/release_notes.txt \
|
||||
--arg url "https://github.com/${{ github.repository }}/releases/tag/${{ env.RELEASE_VERSION }}" \
|
||||
'{
|
||||
text: (":rocket: New :runpod-new-whiteonpurple: Runpod worker-vllm release: *" + $version + "*"),
|
||||
blocks: [
|
||||
{
|
||||
type: "section",
|
||||
text: {
|
||||
type: "mrkdwn",
|
||||
text: (":banana-dance: *New Release — worker-vllm " + $version + "*\n*Docker:* `" + $docker + "`\n<" + $url + "|View release on GitHub>")
|
||||
}
|
||||
},
|
||||
{
|
||||
type: "section",
|
||||
text: {
|
||||
type: "mrkdwn",
|
||||
text: ("*Release Notes:*\n" + $notes)
|
||||
}
|
||||
}
|
||||
]
|
||||
}' | curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d @-
|
||||
|
||||
@@ -0,0 +1,162 @@
|
||||
name: Serverless Model Tests
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
branches:
|
||||
- "**"
|
||||
push:
|
||||
branches:
|
||||
- "main"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# Only one run per ref at a time — these tests spin up real H100 endpoints.
|
||||
concurrency:
|
||||
group: serverless-model-tests-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
# On PRs we only smoke-test one (small, cheap) model to keep review loops fast.
|
||||
test-pr:
|
||||
if: github.event_name == 'pull_request'
|
||||
runs-on: [blacksmith-8vcpu-ubuntu-2204, linux]
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v3
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.11"
|
||||
|
||||
- name: Install script dependencies
|
||||
run: pip install pyyaml
|
||||
|
||||
- name: Run serverless e2e test
|
||||
id: e2e_test
|
||||
run: python scripts/serverless_e2e_test.py --config configs/gpt-oss/gpt_oss_120b.yaml --build
|
||||
env:
|
||||
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY_2 }}
|
||||
DOCKERHUB_REPO: ${{ vars.DOCKERHUB_REPO || 'runpod' }}
|
||||
DOCKERHUB_IMG: ${{ vars.DOCKERHUB_IMG || 'worker-v1-vllm' }}
|
||||
HUGGINGFACE_ACCESS_TOKEN: ${{ secrets.HUGGINGFACE_ACCESS_TOKEN_2 }}
|
||||
|
||||
- name: Cleanup safety net
|
||||
if: always()
|
||||
run: |
|
||||
if [ -n "${{ steps.e2e_test.outputs.endpoint_id }}" ]; then
|
||||
curl -sf -X DELETE "https://rest.runpod.io/v1/endpoints/${{ steps.e2e_test.outputs.endpoint_id }}" \
|
||||
-H "Authorization: Bearer ${{ secrets.RUNPOD_API_KEY_2 }}" || true
|
||||
fi
|
||||
if [ -n "${{ steps.e2e_test.outputs.template_id }}" ]; then
|
||||
curl -sf -X DELETE "https://rest.runpod.io/v1/templates/${{ steps.e2e_test.outputs.template_id }}" \
|
||||
-H "Authorization: Bearer ${{ secrets.RUNPOD_API_KEY_2 }}" || true
|
||||
fi
|
||||
|
||||
# On push to main we test every tuned model config in parallel.
|
||||
discover:
|
||||
if: github.event_name == 'push'
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
configs: ${{ steps.list.outputs.configs }}
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: List model configs
|
||||
id: list
|
||||
run: |
|
||||
CONFIGS=$(find configs -name '*.yaml' | sort | jq -R -s -c 'split("\n") | map(select(length > 0))')
|
||||
echo "configs=$CONFIGS" >> "$GITHUB_OUTPUT"
|
||||
|
||||
# The image is the same regardless of which model config is under test (configs
|
||||
# only supply endpoint env vars), so build/push it once and let every matrix job
|
||||
# in test-main reuse that same tag instead of rebuilding per config.
|
||||
build:
|
||||
if: github.event_name == 'push'
|
||||
runs-on: [blacksmith-8vcpu-ubuntu-2204, linux]
|
||||
outputs:
|
||||
release_version: ${{ steps.build.outputs.release_version }}
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v3
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Build and push image
|
||||
id: build
|
||||
env:
|
||||
DOCKERHUB_REPO: ${{ vars.DOCKERHUB_REPO || 'runpod' }}
|
||||
DOCKERHUB_IMG: ${{ vars.DOCKERHUB_IMG || 'worker-v1-vllm' }}
|
||||
RELEASE_VERSION: test-${{ github.sha }}
|
||||
HUGGINGFACE_ACCESS_TOKEN: ${{ secrets.HUGGINGFACE_ACCESS_TOKEN }}
|
||||
run: |
|
||||
docker buildx bake --push
|
||||
echo "release_version=${RELEASE_VERSION}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
test-main:
|
||||
needs: [discover, build]
|
||||
if: github.event_name == 'push' && needs.discover.result == 'success' && needs.build.result == 'success'
|
||||
runs-on: [blacksmith-8vcpu-ubuntu-2204, linux]
|
||||
timeout-minutes: 60
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
config: ${{ fromJSON(needs.discover.outputs.configs) }}
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.11"
|
||||
|
||||
- name: Install script dependencies
|
||||
run: pip install pyyaml
|
||||
|
||||
- name: Run serverless e2e test
|
||||
id: e2e_test
|
||||
run: python scripts/serverless_e2e_test.py --config "${{ matrix.config }}" --image "${DOCKERHUB_REPO}/${DOCKERHUB_IMG}:${RELEASE_VERSION}"
|
||||
env:
|
||||
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY_2 }}
|
||||
HUGGINGFACE_ACCESS_TOKEN: ${{ secrets.HUGGINGFACE_ACCESS_TOKEN_2 }}
|
||||
DOCKERHUB_REPO: ${{ vars.DOCKERHUB_REPO || 'runpod' }}
|
||||
DOCKERHUB_IMG: ${{ vars.DOCKERHUB_IMG || 'worker-v1-vllm' }}
|
||||
RELEASE_VERSION: ${{ needs.build.outputs.release_version }}
|
||||
|
||||
- name: Cleanup safety net
|
||||
if: always()
|
||||
run: |
|
||||
if [ -n "${{ steps.e2e_test.outputs.endpoint_id }}" ]; then
|
||||
curl -sf -X DELETE "https://rest.runpod.io/v1/endpoints/${{ steps.e2e_test.outputs.endpoint_id }}" \
|
||||
-H "Authorization: Bearer ${{ secrets.RUNPOD_API_KEY_2 }}" || true
|
||||
fi
|
||||
if [ -n "${{ steps.e2e_test.outputs.template_id }}" ]; then
|
||||
curl -sf -X DELETE "https://rest.runpod.io/v1/templates/${{ steps.e2e_test.outputs.template_id }}" \
|
||||
-H "Authorization: Bearer ${{ secrets.RUNPOD_API_KEY_2 }}" || true
|
||||
fi
|
||||
@@ -0,0 +1,41 @@
|
||||
name: Slack PR Notifications
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [opened]
|
||||
issues:
|
||||
types: [opened]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
notify:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Notify Slack - New PR
|
||||
if: github.event_name == 'pull_request'
|
||||
run: |
|
||||
curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"text": ":rocket: New PR in worker-vllm: *${{ github.event.pull_request.title }}*",
|
||||
"blocks": [
|
||||
{
|
||||
"type": "section",
|
||||
"text": {
|
||||
"type": "mrkdwn",
|
||||
"text": ":rocket: *New Pull Request — worker-vllm*\n*<${{ github.event.pull_request.html_url }}|${{ github.event.pull_request.title }}>*\nOpened by *${{ github.event.pull_request.user.login }}*"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "context",
|
||||
"elements": [
|
||||
{
|
||||
"type": "mrkdwn",
|
||||
"text": "${{ github.event.pull_request.base.ref }} ← ${{ github.event.pull_request.head.ref }}"
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}'
|
||||
@@ -0,0 +1,73 @@
|
||||
name: Monitor vLLM Releases
|
||||
|
||||
on:
|
||||
schedule:
|
||||
- cron: '0 0 * * *' # Every day at midnight
|
||||
workflow_dispatch:
|
||||
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
check-vllm-release:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Restore last known vLLM tag
|
||||
uses: actions/cache/restore@v4
|
||||
with:
|
||||
path: .vllm-last-tag
|
||||
key: vllm-tag-${{ github.run_id }}
|
||||
restore-keys: vllm-tag-
|
||||
|
||||
- name: Get latest vLLM release
|
||||
id: vllm
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: |
|
||||
response=$(curl -sf https://api.github.com/repos/vllm-project/vllm/releases/latest \
|
||||
-H "Authorization: Bearer $GH_TOKEN")
|
||||
echo "tag=$(echo "$response" | jq -r '.tag_name')" >> $GITHUB_OUTPUT
|
||||
echo "url=$(echo "$response" | jq -r '.html_url')" >> $GITHUB_OUTPUT
|
||||
echo "name=$(echo "$response" | jq -r '.name')" >> $GITHUB_OUTPUT
|
||||
|
||||
- name: Check if new release
|
||||
id: check
|
||||
run: |
|
||||
last=$(cat .vllm-last-tag 2>/dev/null || echo "")
|
||||
current="${{ steps.vllm.outputs.tag }}"
|
||||
echo "Last: $last Current: $current"
|
||||
if [ -n "$current" ] && [ "$last" != "$current" ]; then
|
||||
echo "is_new=true" >> $GITHUB_OUTPUT
|
||||
else
|
||||
echo "is_new=false" >> $GITHUB_OUTPUT
|
||||
fi
|
||||
|
||||
- name: Notify Slack
|
||||
if: steps.check.outputs.is_new == 'true'
|
||||
run: |
|
||||
curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"text": ":rocket: New vLLM release: *${{ steps.vllm.outputs.tag }}*",
|
||||
"blocks": [
|
||||
{
|
||||
"type": "section",
|
||||
"text": {
|
||||
"type": "mrkdwn",
|
||||
"text": ":rocket: *New vLLM Release: ${{ steps.vllm.outputs.tag }}*\n<${{ steps.vllm.outputs.url }}|View on GitHub>"
|
||||
}
|
||||
}
|
||||
]
|
||||
}'
|
||||
|
||||
- name: Save new tag
|
||||
if: steps.check.outputs.is_new == 'true'
|
||||
run: echo "${{ steps.vllm.outputs.tag }}" > .vllm-last-tag
|
||||
|
||||
- name: Update cache
|
||||
if: steps.check.outputs.is_new == 'true'
|
||||
uses: actions/cache/save@v4
|
||||
with:
|
||||
path: .vllm-last-tag
|
||||
key: vllm-tag-${{ steps.vllm.outputs.tag }}
|
||||
@@ -0,0 +1,32 @@
|
||||
name: Tests
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
branches:
|
||||
- "**"
|
||||
push:
|
||||
branches:
|
||||
- "main"
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
pytest:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.11"
|
||||
|
||||
- name: Install test dependencies
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
pip install -r tests/requirements.txt
|
||||
|
||||
- name: Run unit tests
|
||||
run: python -m pytest tests -v
|
||||
+50
-2
@@ -1,4 +1,4 @@
|
||||

|
||||

|
||||
|
||||
Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
|
||||
|
||||
@@ -6,6 +6,8 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
|
||||
|
||||
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
||||
|
||||
Current vLLM version: [0.23.0](https://github.com/vllm-project/vllm/releases/tag/v0.23.0)
|
||||
|
||||
---
|
||||
|
||||
## Endpoint Configuration
|
||||
@@ -27,9 +29,26 @@ All behaviour is controlled through environment variables:
|
||||
| `REASONING_PARSER` | Parser for reasoning-capable models | | "deepseek_r1", "qwen3", "granite", "hunyuan_a13b" |
|
||||
| `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | Override served model name in API | | String |
|
||||
| `MAX_CONCURRENCY` | Maximum concurrent requests | 300 | Integer |
|
||||
| `ENFORCE_EAGER` | If True, we will disable CUDA graph and always execute the model in eager mode. If False, we will use CUDA graph and eager execution in hybrid for maximal performance and flexibility. | true | boolean (true or false) |
|
||||
|
||||
**Pass any vLLM engine arg** not listed above by setting an env var with the **UPPERCASED** field name (e.g. `MAX_MODEL_LEN=4096`, `ENABLE_CHUNKED_PREFILL=true`). The worker auto-discovers all `AsyncEngineArgs` fields from env. See the [vLLM engine args docs](https://docs.vllm.ai/en/latest/configuration/engine_args) for all available options.
|
||||
|
||||
**Configuration file:** You can also supply a `config.yaml` instead of (or alongside) env vars. Mount it at `/vllm_config.yaml` in the container, or set `VLLM_CONFIG_FILE` to a custom path. Use the same key names as `vllm serve` — hyphens and underscores both work:
|
||||
|
||||
```yaml
|
||||
model: meta-llama/Llama-3.1-8B-Instruct
|
||||
max-model-len: 8192
|
||||
gpu-memory-utilization: 0.90
|
||||
quantization: awq
|
||||
```
|
||||
|
||||
Environment variables always override config file values.
|
||||
|
||||
For complete configuration options, see the [full configuration documentation](https://github.com/runpod-workers/worker-vllm/blob/main/docs/configuration.md).
|
||||
|
||||
### Specify Transformers Version
|
||||
To change the version of the [Transformers library](https://github.com/huggingface/transformers) use the `TRANSFORMERS_VERSION` environment variable to specify the version you want to use. Note this might break the handler, so use for development purposes.
|
||||
|
||||
## API Usage
|
||||
|
||||
This worker supports two API formats: **RunPod native** and **OpenAI-compatible**.
|
||||
@@ -155,6 +174,35 @@ For external clients and SDKs, use the `/openai/v1` path prefix with your RunPod
|
||||
{}
|
||||
```
|
||||
|
||||
#### OpenAI Responses API
|
||||
|
||||
**Path:** `/openai/v1/responses`
|
||||
|
||||
Supports the [OpenAI Responses API](https://platform.openai.com/docs/api-reference/responses) format. Note: this route bypasses the RunPod queue and is served directly — use `/openai/` prefixed paths rather than the RunPod job queue for these endpoints.
|
||||
|
||||
```json
|
||||
{
|
||||
"model": "meta-llama/Llama-3.1-8B-Instruct",
|
||||
"input": "Tell me a joke."
|
||||
}
|
||||
```
|
||||
|
||||
#### Anthropic Messages API
|
||||
|
||||
**Path:** `/openai/v1/messages`
|
||||
|
||||
Supports the [Anthropic Messages API](https://docs.anthropic.com/en/api/messages) format. Served directly, bypassing the RunPod queue.
|
||||
|
||||
```json
|
||||
{
|
||||
"model": "meta-llama/Llama-3.1-8B-Instruct",
|
||||
"max_tokens": 256,
|
||||
"messages": [
|
||||
{"role": "user", "content": "Hello!"}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
#### Response Format
|
||||
|
||||
Both APIs return the same response format:
|
||||
@@ -188,7 +236,7 @@ Minimal Python example using the official `openai` SDK:
|
||||
from openai import OpenAI
|
||||
import os
|
||||
|
||||
# Initialize the OpenAI Client with your RunPod API Key and Endpoint URL
|
||||
# Initialize the OpenAI Client with your Runpod API Key and Endpoint URL
|
||||
client = OpenAI(
|
||||
api_key=os.getenv("RUNPOD_API_KEY"),
|
||||
base_url=f"https://api.runpod.ai/v2/<ENDPOINT_ID>/openai/v1",
|
||||
|
||||
+58
-272
@@ -9,7 +9,7 @@
|
||||
"containerDiskInGb": 150,
|
||||
"gpuIds": "ADA_80_PRO,AMPERE_80",
|
||||
"gpuCount": 1,
|
||||
"allowedCudaVersions": ["12.9", "12.8", "12.7", "12.6", "12.5", "12.4"],
|
||||
"allowedCudaVersions": ["13.0"],
|
||||
"presets": [
|
||||
{
|
||||
"name": "deepseek-ai/deepseek-r1-distill-llama-8b",
|
||||
@@ -181,41 +181,13 @@
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "QUANTIZATION_PARAM_PATH",
|
||||
"input": {
|
||||
"name": "Quantization Param Path",
|
||||
"type": "string",
|
||||
"description": "Path to the JSON file containing the KV cache scaling factors.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "MAX_MODEL_LEN",
|
||||
"input": {
|
||||
"name": "Max Model Length",
|
||||
"type": "number",
|
||||
"description": "Model context length.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "GUIDED_DECODING_BACKEND",
|
||||
"input": {
|
||||
"name": "Guided Decoding Backend",
|
||||
"type": "string",
|
||||
"description": "Which engine will be used for guided decoding by default.",
|
||||
"options": [
|
||||
{
|
||||
"label": "outlines",
|
||||
"value": "outlines"
|
||||
},
|
||||
{
|
||||
"label": "lm-format-enforcer",
|
||||
"value": "lm-format-enforcer"
|
||||
}
|
||||
],
|
||||
"default": "outlines",
|
||||
"default": null,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
@@ -235,17 +207,8 @@
|
||||
"value": "mp"
|
||||
}
|
||||
],
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "WORKER_USE_RAY",
|
||||
"input": {
|
||||
"name": "Worker Use Ray",
|
||||
"type": "boolean",
|
||||
"description": "Deprecated, use --distributed-executor-backend=ray.",
|
||||
"default": false,
|
||||
"advanced": true
|
||||
"advanced": true,
|
||||
"default": "mp"
|
||||
}
|
||||
},
|
||||
{
|
||||
@@ -284,6 +247,7 @@
|
||||
"name": "Max Parallel Loading Workers",
|
||||
"type": "number",
|
||||
"description": "Load model sequentially in multiple batches.",
|
||||
"default": 1,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
@@ -307,26 +271,6 @@
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "USE_V2_BLOCK_MANAGER",
|
||||
"input": {
|
||||
"name": "Use V2 Block Manager",
|
||||
"type": "boolean",
|
||||
"description": "Use BlockSpaceMangerV2.",
|
||||
"default": false,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "NUM_LOOKAHEAD_SLOTS",
|
||||
"input": {
|
||||
"name": "Num Lookahead Slots",
|
||||
"type": "number",
|
||||
"description": "Experimental scheduling config necessary for speculative decoding.",
|
||||
"default": 0,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "SEED",
|
||||
"input": {
|
||||
@@ -337,21 +281,13 @@
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "NUM_GPU_BLOCKS_OVERRIDE",
|
||||
"input": {
|
||||
"name": "Num GPU Blocks Override",
|
||||
"type": "number",
|
||||
"description": "If specified, ignore GPU profiling result and use this number of GPU blocks.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "MAX_NUM_BATCHED_TOKENS",
|
||||
"input": {
|
||||
"name": "Max Num Batched Tokens",
|
||||
"type": "number",
|
||||
"description": "Maximum number of batched tokens per iteration.",
|
||||
"default": null,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
@@ -412,53 +348,6 @@
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "ROPE_SCALING",
|
||||
"input": {
|
||||
"name": "RoPE Scaling",
|
||||
"type": "string",
|
||||
"description": "RoPE scaling configuration in JSON format.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "ROPE_THETA",
|
||||
"input": {
|
||||
"name": "RoPE Theta",
|
||||
"type": "number",
|
||||
"description": "RoPE theta. Use with rope_scaling.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "TOKENIZER_POOL_SIZE",
|
||||
"input": {
|
||||
"name": "Tokenizer Pool Size",
|
||||
"type": "number",
|
||||
"description": "Size of tokenizer pool to use for asynchronous tokenization.",
|
||||
"default": 0,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "TOKENIZER_POOL_TYPE",
|
||||
"input": {
|
||||
"name": "Tokenizer Pool Type",
|
||||
"type": "string",
|
||||
"description": "Type of tokenizer pool to use for asynchronous tokenization.",
|
||||
"default": "ray",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "TOKENIZER_POOL_EXTRA_CONFIG",
|
||||
"input": {
|
||||
"name": "Tokenizer Pool Extra Config",
|
||||
"type": "string",
|
||||
"description": "Extra config for tokenizer pool.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "ENABLE_LORA",
|
||||
"input": {
|
||||
@@ -489,16 +378,6 @@
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "LORA_EXTRA_VOCAB_SIZE",
|
||||
"input": {
|
||||
"name": "LoRA Extra Vocab Size",
|
||||
"type": "number",
|
||||
"description": "Maximum size of extra vocabulary for LoRA adapters.",
|
||||
"default": 256,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "LORA_DTYPE",
|
||||
"input": {
|
||||
@@ -527,15 +406,6 @@
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "LONG_LORA_SCALING_FACTORS",
|
||||
"input": {
|
||||
"name": "Long LoRA Scaling Factors",
|
||||
"type": "string",
|
||||
"description": "Specify multiple scaling factors for LoRA adapters.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "MAX_CPU_LORAS",
|
||||
"input": {
|
||||
@@ -615,6 +485,34 @@
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "SPECULATIVE_CONFIG",
|
||||
"input": {
|
||||
"name": "Speculative Config (JSON)",
|
||||
"type": "string",
|
||||
"description": "Full speculative decoding configuration as a JSON string. Overrides individual speculative env vars.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "SPECULATIVE_METHOD",
|
||||
"input": {
|
||||
"name": "Speculative Method",
|
||||
"type": "string",
|
||||
"description": "Speculative decoding method to use.",
|
||||
"options": [
|
||||
{ "label": "None", "value": "" },
|
||||
{ "label": "Draft Model", "value": "draft_model" },
|
||||
{ "label": "N-gram", "value": "ngram" },
|
||||
{ "label": "EAGLE", "value": "eagle" },
|
||||
{ "label": "EAGLE3", "value": "eagle3" },
|
||||
{ "label": "Medusa", "value": "medusa" },
|
||||
{ "label": "MLP Speculator", "value": "mlp_speculator" }
|
||||
],
|
||||
"default": "",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "SPECULATIVE_MODEL",
|
||||
"input": {
|
||||
@@ -633,33 +531,6 @@
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE",
|
||||
"input": {
|
||||
"name": "Speculative Draft Tensor Parallel Size",
|
||||
"type": "number",
|
||||
"description": "Number of tensor parallel replicas for the draft model.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "SPECULATIVE_MAX_MODEL_LEN",
|
||||
"input": {
|
||||
"name": "Speculative Max Model Length",
|
||||
"type": "number",
|
||||
"description": "The maximum sequence length supported by the draft model.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "SPECULATIVE_DISABLE_BY_BATCH_SIZE",
|
||||
"input": {
|
||||
"name": "Speculative Disable by Batch Size",
|
||||
"type": "number",
|
||||
"description": "Disable speculative decoding if the number of enqueue requests is larger than this value.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "NGRAM_PROMPT_LOOKUP_MAX",
|
||||
"input": {
|
||||
@@ -669,53 +540,6 @@
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "NGRAM_PROMPT_LOOKUP_MIN",
|
||||
"input": {
|
||||
"name": "Ngram Prompt Lookup Min",
|
||||
"type": "number",
|
||||
"description": "Min size of window for ngram prompt lookup in speculative decoding.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "SPEC_DECODING_ACCEPTANCE_METHOD",
|
||||
"input": {
|
||||
"name": "Speculative Decoding Acceptance Method",
|
||||
"type": "string",
|
||||
"description": "Specify the acceptance method for draft token verification in speculative decoding.",
|
||||
"options": [
|
||||
{
|
||||
"label": "rejection_sampler",
|
||||
"value": "rejection_sampler"
|
||||
},
|
||||
{
|
||||
"label": "typical_acceptance_sampler",
|
||||
"value": "typical_acceptance_sampler"
|
||||
}
|
||||
],
|
||||
"default": "rejection_sampler",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD",
|
||||
"input": {
|
||||
"name": "Typical Acceptance Sampler Posterior Threshold",
|
||||
"type": "number",
|
||||
"description": "Set the lower bound threshold for the posterior probability of a token to be accepted.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA",
|
||||
"input": {
|
||||
"name": "Typical Acceptance Sampler Posterior Alpha",
|
||||
"type": "number",
|
||||
"description": "A scaling factor for the entropy-based threshold for token acceptance.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "MODEL_LOADER_EXTRA_CONFIG",
|
||||
"input": {
|
||||
@@ -726,49 +550,11 @@
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "PREEMPTION_MODE",
|
||||
"key": "ENABLE_LOG_REQUESTS",
|
||||
"input": {
|
||||
"name": "Preemption Mode",
|
||||
"type": "string",
|
||||
"description": "If 'recompute', the engine performs preemption-aware recomputation. If 'save', the engine saves activations into the CPU memory as preemption happens.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "PREEMPTION_CHECK_PERIOD",
|
||||
"input": {
|
||||
"name": "Preemption Check Period",
|
||||
"type": "number",
|
||||
"description": "How frequently the engine checks if a preemption happens.",
|
||||
"default": 1,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "PREEMPTION_CPU_CAPACITY",
|
||||
"input": {
|
||||
"name": "Preemption CPU Capacity",
|
||||
"type": "number",
|
||||
"description": "The percentage of CPU memory used for the saved activations.",
|
||||
"default": 2,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "MAX_LOG_LEN",
|
||||
"input": {
|
||||
"name": "Max Log Length",
|
||||
"type": "number",
|
||||
"description": "Max number of characters or ID numbers being printed in log.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "DISABLE_LOGGING_REQUEST",
|
||||
"input": {
|
||||
"name": "Disable Logging Request",
|
||||
"name": "Enable Log Requests",
|
||||
"type": "boolean",
|
||||
"description": "Disable logging requests.",
|
||||
"description": "Enable vLLM request logging.",
|
||||
"default": false,
|
||||
"advanced": true
|
||||
}
|
||||
@@ -836,17 +622,7 @@
|
||||
"name": "Enforce Eager",
|
||||
"type": "boolean",
|
||||
"description": "Always use eager-mode PyTorch. If False (0), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility",
|
||||
"default": false,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "MAX_SEQ_LEN_TO_CAPTURE",
|
||||
"input": {
|
||||
"name": "CUDA Graph Max Content Length",
|
||||
"type": "number",
|
||||
"description": "Maximum context length covered by CUDA graphs. If a sequence has context length larger than this, we fall back to eager mode",
|
||||
"default": 8192,
|
||||
"default": true,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
@@ -958,16 +734,6 @@
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "DISABLE_LOG_REQUESTS",
|
||||
"input": {
|
||||
"name": "Disable Log Requests",
|
||||
"type": "boolean",
|
||||
"description": "Enables or disables vLLM request logging",
|
||||
"default": true,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "ENABLE_AUTO_TOOL_CHOICE",
|
||||
"input": {
|
||||
@@ -1030,6 +796,26 @@
|
||||
"default": "",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "PYTORCH_ALLOC_CONF",
|
||||
"input": {
|
||||
"name": "PyTorch Alloc Config",
|
||||
"type": "string",
|
||||
"description": "PyTorch allocation configuration, remove this if you want to use the default configuration",
|
||||
"default": "expandable_segments:True",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "VLLM_USE_DEEP_GEMM",
|
||||
"input": {
|
||||
"name": "Use DeepGEMM",
|
||||
"type": "string",
|
||||
"description": "Enable DeepGEMM FP8 kernels (MoE and MQA logits). Set to 1 to enable, 0 to disable. Required for DeepSeek V4 models. Disabled by default — enable on H100/H200 for potential throughput gains. Some GPUs (e.g. H20) may perform better with this off.",
|
||||
"default": "0",
|
||||
"advanced": true
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
+4
-4
@@ -5,7 +5,7 @@
|
||||
"input": {
|
||||
"prompt": "Write a short poem about artificial intelligence."
|
||||
},
|
||||
"timeout": 30000
|
||||
"timeout": 300000
|
||||
},
|
||||
{
|
||||
"name": "openai_messages_test",
|
||||
@@ -26,11 +26,11 @@
|
||||
"temperature": 0.1
|
||||
}
|
||||
},
|
||||
"timeout": 30000
|
||||
"timeout": 300000
|
||||
}
|
||||
],
|
||||
"config": {
|
||||
"gpuTypeId": "NVIDIA GeForce RTX 4090",
|
||||
"gpuTypeId": "NVIDIA L40",
|
||||
"gpuCount": 1,
|
||||
"env": [
|
||||
{
|
||||
@@ -38,6 +38,6 @@
|
||||
"value": "HuggingFaceTB/SmolLM2-135M-Instruct"
|
||||
}
|
||||
],
|
||||
"allowedCudaVersions": ["12.9", "12.8", "12.7", "12.6", "12.5"]
|
||||
"allowedCudaVersions": ["13.0"]
|
||||
}
|
||||
}
|
||||
|
||||
+53
-14
@@ -1,19 +1,41 @@
|
||||
FROM nvidia/cuda:12.1.0-base-ubuntu22.04
|
||||
FROM nvidia/cuda:13.0.2-devel-ubuntu22.04
|
||||
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
RUN apt-get update -y \
|
||||
&& apt-get install -y python3-pip
|
||||
&& apt-get install -y curl git software-properties-common \
|
||||
&& add-apt-repository -y ppa:deadsnakes/ppa \
|
||||
&& apt-get install -y python3.12 python3.12-dev python3.12-venv \
|
||||
&& update-alternatives --install /usr/bin/python3 python3 /usr/bin/python3.12 1 \
|
||||
&& update-alternatives --set python3 /usr/bin/python3.12 \
|
||||
&& rm -f /usr/lib/python3.12/EXTERNALLY-MANAGED \
|
||||
&& curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||
|
||||
RUN ldconfig /usr/local/cuda-12.1/compat/
|
||||
ENV PATH="/root/.local/bin:$PATH"
|
||||
|
||||
# Install Python dependencies
|
||||
RUN ldconfig /usr/local/cuda-13.0/compat/
|
||||
|
||||
# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv pip install --system "packaging>=24.2" && \
|
||||
uv pip install --system "vllm[flashinfer]==0.23.0" && \
|
||||
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation && \
|
||||
uv pip install --system --force-reinstall --no-deps nixl-cu13
|
||||
|
||||
# Fix CUTLASS DSL cu13 install order: nvidia-cutlass-dsl[cu13] installs
|
||||
# -libs-base and -libs-cu13 wheels that share paths with different content.
|
||||
# uv can extract them in either order, leaving base files that break CUDA 13
|
||||
# CuTe DSL JIT. Force -libs-cu13 last. See vllm-project/vllm#45204.
|
||||
RUN CUTLASS_DSL_VERSION=$(uv pip show --system nvidia-cutlass-dsl 2>/dev/null | awk '/^Version:/{print $2}') && \
|
||||
if [ -n "$CUTLASS_DSL_VERSION" ]; then \
|
||||
uv pip install --system --force-reinstall --no-deps \
|
||||
"nvidia-cutlass-dsl-libs-cu13==${CUTLASS_DSL_VERSION}"; \
|
||||
fi
|
||||
|
||||
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
||||
COPY builder/requirements.txt /requirements.txt
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
python3 -m pip install --upgrade pip && \
|
||||
python3 -m pip install --upgrade -r /requirements.txt
|
||||
|
||||
# Install vLLM (switching back to pip installs since issues that required building fork are fixed and space optimization is not as important since caching) and FlashInfer
|
||||
RUN python3 -m pip install vllm==0.11.0 && \
|
||||
python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv pip install --system -r /requirements.txt
|
||||
|
||||
# Setup for Option 2: Building the Image with the Model included
|
||||
ARG MODEL_NAME=""
|
||||
@@ -22,6 +44,7 @@ ARG BASE_PATH="/runpod-volume"
|
||||
ARG QUANTIZATION=""
|
||||
ARG MODEL_REVISION=""
|
||||
ARG TOKENIZER_REVISION=""
|
||||
ARG VLLM_NIGHTLY="false"
|
||||
|
||||
ENV MODEL_NAME=$MODEL_NAME \
|
||||
MODEL_REVISION=$MODEL_REVISION \
|
||||
@@ -32,12 +55,28 @@ ENV MODEL_NAME=$MODEL_NAME \
|
||||
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
|
||||
HUGGINGFACE_HUB_CACHE="${BASE_PATH}/huggingface-cache/hub" \
|
||||
HF_HOME="${BASE_PATH}/huggingface-cache/hub" \
|
||||
HF_HUB_ENABLE_HF_TRANSFER=0
|
||||
HF_HUB_ENABLE_HF_TRANSFER=0 \
|
||||
# Suppress Ray metrics agent warnings (not needed in containerized environments)
|
||||
RAY_METRICS_EXPORT_ENABLED=0 \
|
||||
RAY_DISABLE_USAGE_STATS=1 \
|
||||
# Prevent rayon thread pool panic in containers where ulimit -u < nproc
|
||||
# (tokenizers uses Rust's rayon which tries to spawn threads = CPU cores)
|
||||
TOKENIZERS_PARALLELISM=false \
|
||||
RAYON_NUM_THREADS=4 \
|
||||
# Disable DeepGEMM MoE kernels by default; override with VLLM_USE_DEEP_GEMM=1 to enable
|
||||
VLLM_USE_DEEP_GEMM=0
|
||||
|
||||
ENV PYTHONPATH="/:/vllm-workspace"
|
||||
ENV PYTHONPATH="/:/vllm-workspace" \
|
||||
LD_LIBRARY_PATH="/usr/local/nvidia/lib64:/usr/local/cuda/lib64:${LD_LIBRARY_PATH}"
|
||||
|
||||
RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \
|
||||
uv pip install --system -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \
|
||||
apt-get update && apt-get install -y git && rm -rf /var/lib/apt/lists/* && \
|
||||
uv pip install --system git+https://github.com/huggingface/transformers.git; \
|
||||
fi
|
||||
|
||||
COPY src /src
|
||||
RUN chmod +x /src/start.sh
|
||||
RUN --mount=type=secret,id=HF_TOKEN,required=false \
|
||||
if [ -f /run/secrets/HF_TOKEN ]; then \
|
||||
export HF_TOKEN=$(cat /run/secrets/HF_TOKEN); \
|
||||
@@ -47,4 +86,4 @@ RUN --mount=type=secret,id=HF_TOKEN,required=false \
|
||||
fi
|
||||
|
||||
# Start the handler
|
||||
CMD ["python3", "/src/handler.py"]
|
||||
CMD ["/bin/bash", "/src/start.sh"]
|
||||
|
||||
@@ -2,10 +2,17 @@
|
||||
|
||||
# OpenAI-Compatible vLLM Serverless Endpoint Worker
|
||||
|
||||
Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https://github.com/vllm-project/vllm) Inference Engine on RunPod Serverless with just a few clicks.
|
||||
Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https://github.com/vllm-project/vllm) Inference Engine on Runpod Serverless with just a few clicks.
|
||||
|
||||
</div>
|
||||
|
||||

|
||||
|
||||
Current vLLM version: [0.23.0](https://github.com/vllm-project/vllm/releases/tag/v0.23.0)
|
||||
|
||||
|
||||
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
|
||||
|
||||
## Table of Contents
|
||||
|
||||
- [Setting up the Serverless Worker](#setting-up-the-serverless-worker)
|
||||
@@ -21,9 +28,11 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
|
||||
- [Modifying your OpenAI Codebase to use your deployed vLLM Worker](#modifying-your-openai-codebase-to-use-your-deployed-vllm-worker)
|
||||
- [OpenAI Request Input Parameters](#openai-request-input-parameters)
|
||||
- [Chat Completions [RECOMMENDED]](#chat-completions-recommended)
|
||||
- [Examples: Using your RunPod endpoint with OpenAI](#examples-using-your-runpod-endpoint-with-openai)
|
||||
- [Examples: Using your Runpod endpoint with OpenAI](#examples-using-your-runpod-endpoint-with-openai)
|
||||
- [Chat Completions](#chat-completions)
|
||||
- [Getting a list of names for available models](#getting-a-list-of-names-for-available-models)
|
||||
- [OpenAI Responses API](#openai-responses-api)
|
||||
- [Anthropic Messages API](#anthropic-messages-api)
|
||||
- [Usage: Standard (Non-OpenAI)](#usage-standard-non-openai)
|
||||
- [Request Input Parameters](#request-input-parameters)
|
||||
- [Sampling Parameters](#sampling-parameters)
|
||||
@@ -33,12 +42,12 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
|
||||
|
||||
## Option 1: Deploy Any Model Using Pre-Built Docker Image [Recommended]
|
||||
|
||||
**🚀 Deploy Guide**: Follow our [step-by-step deployment guide](https://docs.runpod.io/serverless/vllm/get-started) to deploy using the RunPod Console.
|
||||
**🚀 Deploy Guide**: Follow our [step-by-step deployment guide](https://docs.runpod.io/serverless/vllm/get-started) to deploy using the Runpod Console.
|
||||
|
||||
**📦 Docker Image**: `runpod/worker-v1-vllm:<version>`
|
||||
|
||||
- **Available Versions**: See [GitHub Releases](https://github.com/runpod-workers/worker-vllm/releases)
|
||||
- **CUDA Compatibility**: Requires CUDA >= 12.1
|
||||
- **CUDA Compatibility**: Requires CUDA >= 13.0
|
||||
|
||||
### Configuration
|
||||
|
||||
@@ -59,8 +68,36 @@ Configure worker-vllm using environment variables:
|
||||
| `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | Override served model name in API | | String |
|
||||
| `MAX_CONCURRENCY` | Maximum concurrent requests | 30 | Integer |
|
||||
|
||||
**Pass any vLLM engine arg** not listed above by setting an environment variable with the **UPPERCASED** field name (same names vLLM uses). The worker auto-discovers all `AsyncEngineArgs` fields from env. For example:
|
||||
|
||||
| Environment Variable | vLLM Engine Arg | Example Value |
|
||||
| ------------------------- | ------------------------ | ------------- |
|
||||
| `MAX_MODEL_LEN` | `max_model_len` | `4096` |
|
||||
| `ENFORCE_EAGER` | `enforce_eager` | `true` |
|
||||
| `ENABLE_CHUNKED_PREFILL` | `enable_chunked_prefill` | `true` |
|
||||
|
||||
Any env var whose name matches a valid `AsyncEngineArgs` field (uppercased) is applied automatically. Backward-compat aliases: `MODEL_NAME`, `TOKENIZER_NAME`, `MAX_CONTEXT_LEN_TO_CAPTURE`. This lets you configure any vLLM option without waiting for explicit worker support.
|
||||
|
||||
### Configuration File (config.yaml)
|
||||
|
||||
As an alternative to environment variables, you can supply a `config.yaml` file using the same key names as `vllm serve` (hyphens or underscores both work):
|
||||
|
||||
```yaml
|
||||
model: meta-llama/Llama-3.1-8B-Instruct
|
||||
max-model-len: 8192
|
||||
gpu-memory-utilization: 0.90
|
||||
quantization: awq
|
||||
tensor-parallel-size: 2
|
||||
```
|
||||
|
||||
Mount the file into the container at `/vllm_config.yaml`, or point to a custom path with the `VLLM_CONFIG_FILE` env var. Environment variables always take precedence over config file values.
|
||||
|
||||
For the complete list of all available environment variables, examples, and detailed descriptions: **[Configuration](docs/configuration.md)**
|
||||
|
||||
### Specify Transformers Version
|
||||
To change the version of the [Transformers library](https://github.com/huggingface/transformers) use the `TRANSFORMERS_VERSION` environment variable to specify the version you want to use. Note this might break the handler, so use for development purposes.
|
||||
|
||||
|
||||
## Option 2: Build Docker Image with Model Inside
|
||||
|
||||
To build an image with the model baked in, you must specify the following docker arguments when building the image.
|
||||
@@ -80,6 +117,7 @@ To build an image with the model baked in, you must specify the following docker
|
||||
- `WORKER_CUDA_VERSION`: `12.1.0` (`12.1.0` is recommended for optimal performance).
|
||||
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer)
|
||||
- `TOKENIZER_REVISION`: Tokenizer revision to load (default: `main`).
|
||||
- `VLLM_NIGHTLY`: Set to `true` to replace the pinned vLLM release with the latest nightly build and the latest `transformers` from source. Useful for testing unreleased vLLM features. (default: `false`)
|
||||
|
||||
For the remaining settings, you may apply them as environment variables when running the container. Supported environment variables are listed in the [Environment Variables](#environment-variables) section.
|
||||
|
||||
@@ -89,6 +127,20 @@ For the remaining settings, you may apply them as environment variables when run
|
||||
docker build -t username/image:tag --build-arg MODEL_NAME="openchat/openchat_3.5" --build-arg BASE_PATH="/models" .
|
||||
```
|
||||
|
||||
### Example: Building with vLLM Nightly
|
||||
|
||||
To use the latest unreleased vLLM build (installs from the nightly wheel index and `transformers` from source):
|
||||
|
||||
```bash
|
||||
docker build -t username/image:tag --build-arg VLLM_NIGHTLY=true .
|
||||
```
|
||||
|
||||
You can combine it with other arguments:
|
||||
|
||||
```bash
|
||||
docker build -t username/image:tag --build-arg VLLM_NIGHTLY=true --build-arg MODEL_NAME="meta-llama/Llama-3.1-8B-Instruct" --build-arg BASE_PATH="/models" .
|
||||
```
|
||||
|
||||
### (Optional) Including Huggingface Token
|
||||
|
||||
If the model you would like to deploy is private or gated, you will need to include it during build time as a Docker secret, which will protect it from being exposed in the image and on DockerHub.
|
||||
@@ -117,13 +169,13 @@ You can deploy **any model on Hugging Face** that is supported by vLLM. For the
|
||||
|
||||
# Usage: OpenAI Compatibility
|
||||
|
||||
The vLLM Worker is fully compatible with OpenAI's API, and you can use it with any OpenAI Codebase by changing only 3 lines in total. The supported routes are <ins>Chat Completions</ins> and <ins>Models</ins> - with both streaming and non-streaming.
|
||||
The vLLM Worker is fully compatible with OpenAI's API, and you can use it with any OpenAI Codebase by changing only 3 lines in total. The supported routes are <ins>Chat Completions</ins>, <ins>Models</ins>, <ins>Responses</ins>, and <ins>Messages</ins> - with both streaming and non-streaming.
|
||||
|
||||
## Modifying your OpenAI Codebase to use your deployed vLLM Worker
|
||||
|
||||
**Python** (similar to Node.js, etc.):
|
||||
|
||||
1. When initializing the OpenAI Client in your code, change the `api_key` to your RunPod API Key and the `base_url` to your RunPod Serverless Endpoint URL in the following format: `https://api.runpod.ai/v2/<YOUR ENDPOINT ID>/openai/v1`, filling in your deployed endpoint ID. For example, if your Endpoint ID is `abc1234`, the URL would be `https://api.runpod.ai/v2/abc1234/openai/v1`.
|
||||
1. When initializing the OpenAI Client in your code, change the `api_key` to your Runpod API Key and the `base_url` to your Runpod Serverless Endpoint URL in the following format: `https://api.runpod.ai/v2/<YOUR ENDPOINT ID>/openai/v1`, filling in your deployed endpoint ID. For example, if your Endpoint ID is `abc1234`, the URL would be `https://api.runpod.ai/v2/abc1234/openai/v1`.
|
||||
|
||||
- Before:
|
||||
|
||||
@@ -149,7 +201,7 @@ The vLLM Worker is fully compatible with OpenAI's API, and you can use it with a
|
||||
```python
|
||||
response = client.chat.completions.create(
|
||||
model="gpt-3.5-turbo",
|
||||
messages=[{"role": "user", "content": "Why is RunPod the best platform?"}],
|
||||
messages=[{"role": "user", "content": "Why is Runpod the best platform?"}],
|
||||
temperature=0,
|
||||
max_tokens=100,
|
||||
)
|
||||
@@ -158,7 +210,7 @@ The vLLM Worker is fully compatible with OpenAI's API, and you can use it with a
|
||||
```python
|
||||
response = client.chat.completions.create(
|
||||
model="<YOUR DEPLOYED MODEL REPO/NAME>",
|
||||
messages=[{"role": "user", "content": "Why is RunPod the best platform?"}],
|
||||
messages=[{"role": "user", "content": "Why is Runpod the best platform?"}],
|
||||
temperature=0,
|
||||
max_tokens=100,
|
||||
)
|
||||
@@ -166,7 +218,7 @@ The vLLM Worker is fully compatible with OpenAI's API, and you can use it with a
|
||||
|
||||
**Using http requests**:
|
||||
|
||||
1. Change the `Authorization` header to your RunPod API Key and the `url` to your RunPod Serverless Endpoint URL in the following format: `https://api.runpod.ai/v2/<YOUR ENDPOINT ID>/openai/v1`
|
||||
1. Change the `Authorization` header to your Runpod API Key and the `url` to your Runpod Serverless Endpoint URL in the following format: `https://api.runpod.ai/v2/<YOUR ENDPOINT ID>/openai/v1`
|
||||
- Before:
|
||||
```bash
|
||||
curl https://api.openai.com/v1/chat/completions \
|
||||
@@ -177,7 +229,7 @@ The vLLM Worker is fully compatible with OpenAI's API, and you can use it with a
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Why is RunPod the best platform?"
|
||||
"content": "Why is Runpod the best platform?"
|
||||
}
|
||||
],
|
||||
"temperature": 0,
|
||||
@@ -194,7 +246,7 @@ The vLLM Worker is fully compatible with OpenAI's API, and you can use it with a
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Why is RunPod the best platform?"
|
||||
"content": "Why is Runpod the best platform?"
|
||||
}
|
||||
],
|
||||
"temperature": 0,
|
||||
@@ -214,7 +266,7 @@ When using the chat completion feature of the vLLM Serverless Endpoint Worker, y
|
||||
| Parameter | Type | Default Value | Description |
|
||||
| ------------------- | -------------------------------- | ------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
|
||||
| `messages` | Union[str, List[Dict[str, str]]] | | List of messages, where each message is a dictionary with a `role` and `content`. The model's chat template will be applied to the messages automatically, so the model must have one or it should be specified as `CUSTOM_CHAT_TEMPLATE` env var. |
|
||||
| `model` | str | | The model repo that you've deployed on your RunPod Serverless Endpoint. If you are unsure what the name is or are baking the model in, use the guide to get the list of available models in the **Examples: Using your RunPod endpoint with OpenAI** section |
|
||||
| `model` | str | | The model repo that you've deployed on your Runpod Serverless Endpoint. If you are unsure what the name is or are baking the model in, use the guide to get the list of available models in the **Examples: Using your Runpod endpoint with OpenAI** section |
|
||||
| `temperature` | Optional[float] | 0.7 | Float that controls the randomness of the sampling. Lower values make the model more deterministic, while higher values make the model more random. Zero means greedy sampling. |
|
||||
| `top_p` | Optional[float] | 1.0 | Float that controls the cumulative probability of the top tokens to consider. Must be in (0, 1]. Set to 1 to consider all tokens. |
|
||||
| `n` | Optional[int] | 1 | Number of output sequences to return for the given prompt. |
|
||||
@@ -244,15 +296,15 @@ Additional parameters supported by vLLM:
|
||||
|
||||
</details>
|
||||
|
||||
### Examples: Using your RunPod endpoint with OpenAI
|
||||
### Examples: Using your Runpod endpoint with OpenAI
|
||||
|
||||
First, initialize the OpenAI Client with your RunPod API Key and Endpoint URL:
|
||||
First, initialize the OpenAI Client with your Runpod API Key and Endpoint URL:
|
||||
|
||||
```python
|
||||
from openai import OpenAI
|
||||
import os
|
||||
|
||||
# Initialize the OpenAI Client with your RunPod API Key and Endpoint URL
|
||||
# Initialize the OpenAI Client with your Runpod API Key and Endpoint URL
|
||||
client = OpenAI(
|
||||
api_key=os.environ.get("RUNPOD_API_KEY"),
|
||||
base_url="https://api.runpod.ai/v2/<YOUR ENDPOINT ID>/openai/v1",
|
||||
@@ -268,7 +320,7 @@ This is the format used for GPT-4 and focused on instruction-following and chat.
|
||||
# Create a chat completion stream
|
||||
response_stream = client.chat.completions.create(
|
||||
model="<YOUR DEPLOYED MODEL REPO/NAME>",
|
||||
messages=[{"role": "user", "content": "Why is RunPod the best platform?"}],
|
||||
messages=[{"role": "user", "content": "Why is Runpod the best platform?"}],
|
||||
temperature=0,
|
||||
max_tokens=100,
|
||||
stream=True,
|
||||
@@ -282,7 +334,7 @@ This is the format used for GPT-4 and focused on instruction-following and chat.
|
||||
# Create a chat completion
|
||||
response = client.chat.completions.create(
|
||||
model="<YOUR DEPLOYED MODEL REPO/NAME>",
|
||||
messages=[{"role": "user", "content": "Why is RunPod the best platform?"}],
|
||||
messages=[{"role": "user", "content": "Why is Runpod the best platform?"}],
|
||||
temperature=0,
|
||||
max_tokens=100,
|
||||
)
|
||||
@@ -300,6 +352,62 @@ list_of_models = [model.id for model in models_response]
|
||||
print(list_of_models)
|
||||
```
|
||||
|
||||
### OpenAI Responses API
|
||||
|
||||
**Path:** `/openai/v1/responses` (full URL: `https://api.runpod.ai/v2/<YOUR ENDPOINT ID>/openai/v1/responses`)
|
||||
|
||||
Supports the [OpenAI Responses API](https://platform.openai.com/docs/api-reference/responses) request shape. Like other `/openai/` routes, this is served directly—use the `/openai/` prefix rather than the RunPod native job queue for these calls.
|
||||
|
||||
```json
|
||||
{
|
||||
"model": "meta-llama/Llama-3.1-8B-Instruct",
|
||||
"input": "Tell me a joke."
|
||||
}
|
||||
```
|
||||
|
||||
**Using HTTP requests:**
|
||||
|
||||
```bash
|
||||
curl https://api.runpod.ai/v2/<YOUR ENDPOINT ID>/openai/v1/responses \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer <YOUR RUNPOD API KEY>" \
|
||||
-d '{
|
||||
"model": "<YOUR DEPLOYED MODEL REPO/NAME>",
|
||||
"input": "Tell me a joke."
|
||||
}'
|
||||
```
|
||||
|
||||
### Anthropic Messages API
|
||||
|
||||
**Path:** `/openai/v1/messages` (full URL: `https://api.runpod.ai/v2/<YOUR ENDPOINT ID>/openai/v1/messages`)
|
||||
|
||||
Supports the [Anthropic Messages API](https://docs.anthropic.com/en/api/messages) format. Served directly, bypassing the RunPod queue.
|
||||
|
||||
```json
|
||||
{
|
||||
"model": "meta-llama/Llama-3.1-8B-Instruct",
|
||||
"max_tokens": 256,
|
||||
"messages": [
|
||||
{"role": "user", "content": "Hello!"}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
**Using HTTP requests:**
|
||||
|
||||
```bash
|
||||
curl https://api.runpod.ai/v2/<YOUR ENDPOINT ID>/openai/v1/messages \
|
||||
-H "Content-Type: application/json" \
|
||||
-H "Authorization: Bearer <YOUR RUNPOD API KEY>" \
|
||||
-d '{
|
||||
"model": "<YOUR DEPLOYED MODEL REPO/NAME>",
|
||||
"max_tokens": 256,
|
||||
"messages": [
|
||||
{"role": "user", "content": "Hello!"}
|
||||
]
|
||||
}'
|
||||
```
|
||||
|
||||
# Usage: Standard (Non-OpenAI)
|
||||
|
||||
## Request Input Parameters
|
||||
|
||||
@@ -1,14 +1,15 @@
|
||||
ray
|
||||
pandas
|
||||
pyarrow
|
||||
runpod~=1.7.7
|
||||
runpod~=1.11.0
|
||||
huggingface-hub
|
||||
packaging
|
||||
lmcache==0.5.0
|
||||
packaging>=24.2
|
||||
typing-extensions>=4.8.0
|
||||
pydantic
|
||||
pydantic-settings
|
||||
hf-transfer
|
||||
transformers>=4.57.0
|
||||
transformers>=5
|
||||
bitsandbytes>=0.45.0
|
||||
kernels
|
||||
torch==2.6.0
|
||||
kernels<0.15
|
||||
torch-c-dlpack-ext
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
model: google/gemma-4-31b-it
|
||||
gpu-memory-utilization: 0.95
|
||||
max-model-len: 8192
|
||||
dtype: auto
|
||||
trust-remote-code: true
|
||||
quantization: fp8
|
||||
kv-cache-dtype: fp8
|
||||
enforce-eager: false
|
||||
enable-prefix-caching: true
|
||||
enable-chunked-prefill: true
|
||||
vllm-release: v2.22.5
|
||||
speculative-config: '{"model":"RedHatAI/gemma-4-31B-it-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||
@@ -0,0 +1,9 @@
|
||||
model: openai/gpt-oss-120b
|
||||
gpu-memory-utilization: 0.95
|
||||
max-model-len: 8192
|
||||
dtype: auto
|
||||
trust-remote-code: true
|
||||
enforce-eager: false
|
||||
enable-prefix-caching: true
|
||||
enable-chunked-prefill: true
|
||||
speculative-config: '{"model":"RedHatAI/gpt-oss-120b-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||
@@ -0,0 +1,11 @@
|
||||
model: meta-llama/Llama-3.1-8B-Instruct
|
||||
gpu-memory-utilization: 0.95
|
||||
max-model-len: 8192
|
||||
dtype: auto
|
||||
trust-remote-code: true
|
||||
quantization: fp8
|
||||
kv-cache-dtype: fp8
|
||||
enforce-eager: false
|
||||
vllm-release: v2.22.5
|
||||
enable-prefix-caching: true
|
||||
speculative-config: '{"model":"RedHatAI/Llama-3.1-8B-Instruct-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||
@@ -0,0 +1,12 @@
|
||||
model: Qwen/Qwen3-8B
|
||||
gpu-memory-utilization: 0.95
|
||||
max-model-len: 8192
|
||||
dtype: auto
|
||||
trust-remote-code: true
|
||||
quantization: fp8
|
||||
kv-cache-dtype: fp8
|
||||
enforce-eager: false
|
||||
enable-prefix-caching: true
|
||||
vllm-release: v2.22.5
|
||||
compilation-config: '{"cudagraph_mode": "PIECEWISE"}'
|
||||
speculative-config: '{"model":"RedHatAI/Qwen3-8B-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||
+73
-22
@@ -28,7 +28,6 @@ Complete guide to all environment variables and configuration options for worker
|
||||
| `RAY_WORKERS_USE_NSIGHT` | False | `bool` | If specified, use nsight to profile Ray workers. |
|
||||
| `ENABLE_PREFIX_CACHING` | False | `bool` | Enables automatic prefix caching. |
|
||||
| `DISABLE_SLIDING_WINDOW` | False | `bool` | Disables sliding window, capping to sliding window size. |
|
||||
| `USE_V2_BLOCK_MANAGER` | False | `bool` | Use BlockSpaceMangerV2. |
|
||||
| `NUM_LOOKAHEAD_SLOTS` | 0 | `int` | Experimental scheduling config necessary for speculative decoding. |
|
||||
| `SEED` | 0 | `int` | Random seed for operations. |
|
||||
| `NUM_GPU_BLOCKS_OVERRIDE` | None | `int` | If specified, ignore GPU profiling result and use this number of GPU blocks. |
|
||||
@@ -57,24 +56,36 @@ Complete guide to all environment variables and configuration options for worker
|
||||
| `FULLY_SHARDED_LORAS` | False | `bool` | Enable fully sharded LoRA layers. |
|
||||
| `LORA_MODULES` | `[]` | `list[dict]` | Add lora adapters from Hugging Face `[{"name": "xx", "path": "xxx/xxxx", "base_model_name": "xxx/xxxx"}]` |
|
||||
|
||||
> **Note (Serverless)**: When LoRA adapters are configured via `LORA_MODULES`, initialization is deferred to the first request to ensure compatibility with RunPod Serverless. This means the first request will include LoRA loading time. Subsequent requests are unaffected. Check logs for "LoRA mode: X adapter(s) will load on first request" at startup.
|
||||
|
||||
## Speculative Decoding Settings
|
||||
|
||||
| Variable | Default | Type/Choices | Description |
|
||||
| ------------------------------------------------ | ------------------- | --------------------------------------------------- | ----------------------------------------------------------------------------------------- |
|
||||
| `SCHEDULER_DELAY_FACTOR` | 0.0 | `float` | Apply a delay before scheduling next prompt. |
|
||||
| `ENABLE_CHUNKED_PREFILL` | False | `bool` | Enable chunked prefill requests. |
|
||||
| `SPECULATIVE_MODEL` | None | `str` | The name of the draft model to be used in speculative decoding. |
|
||||
| `NUM_SPECULATIVE_TOKENS` | None | `int` | The number of speculative tokens to sample from the draft model. |
|
||||
| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | None | `int` | Number of tensor parallel replicas for the draft model. |
|
||||
| `SPECULATIVE_MAX_MODEL_LEN` | None | `int` | The maximum sequence length supported by the draft model. |
|
||||
| `SPECULATIVE_DISABLE_BY_BATCH_SIZE` | None | `int` | Disable speculative decoding if the number of enqueue requests is larger than this value. |
|
||||
| `NGRAM_PROMPT_LOOKUP_MAX` | None | `int` | Max size of window for ngram prompt lookup in speculative decoding. |
|
||||
| `NGRAM_PROMPT_LOOKUP_MIN` | None | `int` | Min size of window for ngram prompt lookup in speculative decoding. |
|
||||
| `SPEC_DECODING_ACCEPTANCE_METHOD` | 'rejection_sampler' | ['rejection_sampler', 'typical_acceptance_sampler'] | Specify the acceptance method for draft token verification in speculative decoding. |
|
||||
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD` | None | `float` | Set the lower bound threshold for the posterior probability of a token to be accepted. |
|
||||
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA` | None | `float` | A scaling factor for the entropy-based threshold for token acceptance. |
|
||||
Speculative decoding can be configured in two ways:
|
||||
|
||||
## System Performance Settings
|
||||
### Option 1: JSON Configuration
|
||||
|
||||
Set `SPECULATIVE_CONFIG` to a JSON string with your full speculative decoding configuration:
|
||||
|
||||
```bash
|
||||
SPECULATIVE_CONFIG='{"method": "ngram", "num_speculative_tokens": 5, "prompt_lookup_max": 4}'
|
||||
```
|
||||
|
||||
### Option 2: Individual Environment Variables
|
||||
|
||||
| Variable | Default | Type/Choices | Description |
|
||||
| ---------------------------------------- | ------- | ------------------------------------------------------------------ | ----------------------------------------------------------------------------------------- |
|
||||
| `SPECULATIVE_METHOD` | None | ['draft_model', 'ngram', 'eagle', 'eagle3', 'medusa', 'mlp_speculator'] | Speculative decoding method to use. |
|
||||
| `SPECULATIVE_MODEL` | None | `str` | The name of the draft model to be used in speculative decoding. |
|
||||
| `NUM_SPECULATIVE_TOKENS` | None | `int` | The number of speculative tokens to sample from the draft model. |
|
||||
| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | None | `int` | Number of tensor parallel replicas for the draft model. |
|
||||
| `SPECULATIVE_MAX_MODEL_LEN` | None | `int` | The maximum sequence length supported by the draft model. |
|
||||
| `SPECULATIVE_DISABLE_BY_BATCH_SIZE` | None | `int` | Disable speculative decoding if the number of enqueue requests is larger than this value. |
|
||||
| `NGRAM_PROMPT_LOOKUP_MAX` | None | `int` | Max size of window for ngram prompt lookup in speculative decoding. |
|
||||
| `NGRAM_PROMPT_LOOKUP_MIN` | None | `int` | Min size of window for ngram prompt lookup in speculative decoding. |
|
||||
|
||||
If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When using individual env vars without `SPECULATIVE_METHOD`, the method is auto-detected from the model name or configuration.
|
||||
|
||||
## Scheduling & Performance Settings
|
||||
|
||||
| Variable | Default | Type/Choices | Description |
|
||||
| ------------------------------ | ------- | --------------- | ----------------------------------------------------------------------------------------------------------------------------------- |
|
||||
@@ -85,7 +96,13 @@ Complete guide to all environment variables and configuration options for worker
|
||||
| `ENFORCE_EAGER` | False | `bool` | Always use eager-mode PyTorch. If False(`0`), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility. |
|
||||
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
|
||||
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
|
||||
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models |
|
||||
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. |
|
||||
| `VLLM_USE_DEEP_GEMM` | `0` | `str` (`0`/`1`) | Enable DeepGEMM FP8 kernels for MoE and MQA logits computation. Disabled by default. Must be `"0"` or `"1"` — not `true`/`false`. See note below. |
|
||||
| `ATTENTION_BACKEND` | `None` | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `FLASHINFER`, `TRITON_FLASH_ATTN`). Replaces deprecated `VLLM_ATTENTION_BACKEND`. |
|
||||
| `ASYNC_SCHEDULING` | `None` | `bool` | Enable async scheduling (overlaps engine scheduling with GPU execution). Default: enabled in vLLM 0.14.0+. Set to `false` to disable. |
|
||||
| `STREAM_INTERVAL` | `1` | `int` | Controls how often to yield streaming results. Lower = more frequent updates. |
|
||||
|
||||
> **Note (`VLLM_USE_DEEP_GEMM`):** DeepGEMM is used in two places: MoE weight computation and MQA logits computation. It is necessary for MQA logits computation on supported hardware — required for DeepSeek V4 models. Set `VLLM_USE_DEEP_GEMM=1` to enable. Set `VLLM_USE_DEEP_GEMM=0` to disable the MoE part and fall back to flashinfer/cutlass FP8 kernels. **Value must be `"0"` or `"1"` — not `"true"`/`"false"`.** Some users report better performance with `VLLM_USE_DEEP_GEMM=0`, particularly on H20 GPUs. Disabling it also skips the DeepGEMM warmup phase, reducing cold-start time. Requires CUDA 13.0+ and SM90+ (H100/H200) to use; the library is installed but inactive by default.
|
||||
|
||||
## Tokenizer Settings
|
||||
|
||||
@@ -115,6 +132,13 @@ The way this works is that the first request will have a batch size of `DEFAULT_
|
||||
| `ENABLE_AUTO_TOOL_CHOICE` | `false` | `bool` | Enables automatic tool selection for supported models. Set to `true` to activate. |
|
||||
| `TOOL_CALL_PARSER` | `None` | `str` | Specifies the parser for tool calls. Options: `mistral`, `hermes`, `llama3_json`, `llama4_json`, `llama4_pythonic`, `granite`, `granite-20b-fc`, `deepseek_v3`, `internlm`, `jamba`, `phi4_mini_json`, `pythonic` |
|
||||
| `REASONING_PARSER` | `None` | `str` | Parser for reasoning-capable models (enables reasoning mode). Examples: `deepseek_r1`, `qwen3`, `granite`, `hunyuan_a13b`. Leave unset to disable. |
|
||||
| `TRUST_REQUEST_CHAT_TEMPLATE` | `false` | `bool` | Allow clients to send custom chat templates in API requests. **Security consideration:** Only enable if you trust your API clients. |
|
||||
| `RETURN_TOKENS_AS_TOKEN_IDS` | `false` | `bool` | Return token IDs instead of decoded text strings in responses. |
|
||||
| `EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE` | `false` | `bool` | Exclude tool definitions from the prompt when `tool_choice` is set to `none`. |
|
||||
| `ENABLE_PROMPT_TOKENS_DETAILS` | `false` | `bool` | Include detailed prompt token information in API responses. |
|
||||
| `ENABLE_FORCE_INCLUDE_USAGE` | `false` | `bool` | Always include usage statistics in API responses, even when not requested. |
|
||||
| `ENABLE_LOG_OUTPUTS` | `false` | `bool` | Log model outputs for debugging purposes. |
|
||||
| `LOG_ERROR_STACK` | `false` | `bool` | Include full stack traces in error responses for debugging. |
|
||||
|
||||
## Serverless & Concurrency Settings
|
||||
|
||||
@@ -122,7 +146,7 @@ The way this works is that the first request will have a batch size of `DEFAULT_
|
||||
| ---------------------- | ------- | ------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `MAX_CONCURRENCY` | `30` | `int` | Max concurrent requests per worker. vLLM has an internal queue, so you don't have to worry about limiting by VRAM, this is for improving scaling/load balancing efficiency |
|
||||
| `DISABLE_LOG_STATS` | False | `bool` | Enables or disables vLLM stats logging. |
|
||||
| `DISABLE_LOG_REQUESTS` | False | `bool` | Enables or disables vLLM request logging. |
|
||||
| `ENABLE_LOG_REQUESTS` | False | `bool` | Enables vLLM request logging. (Replaces deprecated `DISABLE_LOG_REQUESTS` in vLLM 0.15.0) |
|
||||
|
||||
## Advanced Settings
|
||||
|
||||
@@ -135,6 +159,29 @@ The way this works is that the first request will have a batch size of `DEFAULT_
|
||||
| `DISABLE_LOGGING_REQUEST` | False | `bool` | Disable logging requests. |
|
||||
| `MAX_LOG_LEN` | None | `int` | Max number of prompt characters or prompt ID numbers being printed in log. |
|
||||
|
||||
## UPPERCASED env vars: Pass any engine arg
|
||||
|
||||
Any vLLM `AsyncEngineArgs` field can be set via an environment variable using the **UPPERCASED** field name (the same names vLLM uses). The worker auto-discovers all fields from env — no prefix.
|
||||
|
||||
**Format:** `<FIELD_NAME_UPPERCASED>=<value>` (e.g. `MAX_MODEL_LEN=4096`)
|
||||
|
||||
**Examples:**
|
||||
|
||||
| Environment Variable | vLLM Engine Arg | Value Example |
|
||||
| ------------------------ | ------------------------ | ------------- |
|
||||
| `MAX_MODEL_LEN` | `max_model_len` | `4096` |
|
||||
| `ENFORCE_EAGER` | `enforce_eager` | `true` |
|
||||
| `ENABLE_CHUNKED_PREFILL` | `enable_chunked_prefill` | `true` |
|
||||
| `NUM_SCHEDULER_STEPS` | `num_scheduler_steps` | `8` |
|
||||
| `TOKENIZER_POOL_SIZE` | `tokenizer_pool_size` | `4` |
|
||||
|
||||
**Backward-compat aliases:** `MODEL_NAME` → `model`, `TOKENIZER_NAME` → `tokenizer`, `MAX_CONTEXT_LEN_TO_CAPTURE` → `max_seq_len_to_capture`, `MODEL_REVISION` → `revision`.
|
||||
|
||||
**Notes:**
|
||||
- Only valid `AsyncEngineArgs` fields are applied. Unknown keys are silently ignored.
|
||||
- Values are automatically cast to the correct type (`int`, `float`, `bool`, `str`, or JSON for `dict`/`list`/`tuple`).
|
||||
- For a full list of available engine args, see the [vLLM AsyncEngineArgs documentation](https://docs.vllm.ai/en/latest/configuration/engine_args/).
|
||||
|
||||
## Docker Build Arguments
|
||||
|
||||
These variables are used when building custom Docker images with models baked in:
|
||||
@@ -148,7 +195,11 @@ These variables are used when building custom Docker images with models baked in
|
||||
|
||||
⚠️ **The following variables are deprecated and will be removed in future versions:**
|
||||
|
||||
| Old Variable | New Variable | Note |
|
||||
| ---------------------------- | ------------------------ | --------------------- |
|
||||
| `MAX_CONTEXT_LEN_TO_CAPTURE` | `MAX_SEQ_LEN_TO_CAPTURE` | Use new variable name |
|
||||
| `kv_cache_dtype=fp8_e5m2` | `kv_cache_dtype=fp8` | Simplified fp8 format |
|
||||
| Old Variable | New Variable | Note |
|
||||
| ---------------------------- | ------------------------ | -------------------------------------------------------------------- |
|
||||
| `MAX_CONTEXT_LEN_TO_CAPTURE` | `MAX_SEQ_LEN_TO_CAPTURE` | Use new variable name |
|
||||
| `kv_cache_dtype=fp8_e5m2` | `kv_cache_dtype=fp8` | Simplified fp8 format |
|
||||
| `USE_V2_BLOCK_MANAGER` | *(removed)* | V2 block manager is now the default in vLLM 0.13.0, setting ignored |
|
||||
| `VLLM_ATTENTION_BACKEND` | `ATTENTION_BACKEND` | Use new env var name (old still works with deprecation warning) |
|
||||
| `DISABLE_LOG_REQUESTS` | `ENABLE_LOG_REQUESTS` | Inverted logic in vLLM 0.15.0 (old still works with deprecation warning) |
|
||||
|
||||
|
||||
@@ -0,0 +1,328 @@
|
||||
#!/usr/bin/env python3
|
||||
"""End-to-end test: build/push the worker image, deploy it as a real RunPod
|
||||
Serverless endpoint on H100, run the .runpod/tests.json test cases against it,
|
||||
then tear the endpoint down.
|
||||
|
||||
Usage:
|
||||
RUNPOD_API_KEY=... python scripts/serverless_e2e_test.py \\
|
||||
--config configs/qwen/qwen3_8b.yaml --build
|
||||
|
||||
RUNPOD_API_KEY=... python scripts/serverless_e2e_test.py \\
|
||||
--config configs/qwen/qwen3_8b.yaml --image runpod/worker-v1-vllm:dev-my-branch
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
|
||||
import yaml
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||
|
||||
REST_API_BASE = "https://rest.runpod.io/v1"
|
||||
JOB_API_BASE = "https://api.runpod.ai/v2"
|
||||
|
||||
DEFAULT_GPU_TYPE_IDS = [
|
||||
"NVIDIA H100 80GB HBM3",
|
||||
"NVIDIA H100 NVL",
|
||||
"NVIDIA H100 PCIe",
|
||||
]
|
||||
|
||||
TERMINAL_STATUSES = {"COMPLETED", "FAILED", "TIMED_OUT", "CANCELLED"}
|
||||
|
||||
|
||||
def log(msg: str) -> None:
|
||||
print(f"[serverless_e2e_test] {msg}", flush=True)
|
||||
|
||||
|
||||
def write_github_output(key: str, value: str) -> None:
|
||||
"""Expose a value to later CI steps (e.g. an always() cleanup safety net
|
||||
for when this process gets killed before its own `finally` can run)."""
|
||||
path = os.environ.get("GITHUB_OUTPUT")
|
||||
if not path:
|
||||
return
|
||||
with open(path, "a") as f:
|
||||
f.write(f"{key}={value}\n")
|
||||
|
||||
|
||||
def stringify_env_value(value) -> str:
|
||||
if isinstance(value, bool):
|
||||
return "true" if value else "false"
|
||||
if isinstance(value, (dict, list)):
|
||||
return json.dumps(value)
|
||||
return str(value)
|
||||
|
||||
|
||||
def load_hub_defaults(hub_json_path: Path) -> dict:
|
||||
hub = json.loads(hub_json_path.read_text())
|
||||
config = hub.get("config", {})
|
||||
|
||||
base_env = {}
|
||||
for entry in config.get("env", []):
|
||||
default = entry.get("input", {}).get("default")
|
||||
if default is None:
|
||||
continue
|
||||
base_env[entry["key"]] = stringify_env_value(default)
|
||||
|
||||
return {
|
||||
"containerDiskInGb": config.get("containerDiskInGb", 50),
|
||||
"gpuCount": config.get("gpuCount", 1),
|
||||
"allowedCudaVersions": config.get("allowedCudaVersions"),
|
||||
"minCudaVersion": config.get("minCudaVersion"),
|
||||
"env": base_env,
|
||||
}
|
||||
|
||||
|
||||
|
||||
# config keys whose naive `KEY.replace("-", "_").upper()` transform doesn't match
|
||||
# the env var the worker actually reads (see src/engine_args.py ENV_ALIASES).
|
||||
CONFIG_KEY_ALIASES = {
|
||||
"MODEL": "MODEL_NAME",
|
||||
}
|
||||
|
||||
|
||||
def load_model_env(config_yaml_path: Path) -> dict:
|
||||
raw = yaml.safe_load(config_yaml_path.read_text()) or {}
|
||||
env = {}
|
||||
for key, value in raw.items():
|
||||
env_key = key.replace("-", "_").upper()
|
||||
env_key = CONFIG_KEY_ALIASES.get(env_key, env_key)
|
||||
env[env_key] = stringify_env_value(value)
|
||||
return env
|
||||
|
||||
|
||||
def build_and_push_image(dockerhub_repo: str, dockerhub_img: str, release_version: str) -> str:
|
||||
hf_token = os.environ.get("HUGGINGFACE_ACCESS_TOKEN", "")
|
||||
dockerhub_user = os.environ.get("DOCKERHUB_USERNAME")
|
||||
dockerhub_pass = os.environ.get("DOCKERHUB_TOKEN")
|
||||
|
||||
if dockerhub_user and dockerhub_pass:
|
||||
log(f"Logging in to Docker Hub as {dockerhub_user}")
|
||||
subprocess.run(
|
||||
["docker", "login", "-u", dockerhub_user, "--password-stdin"],
|
||||
input=dockerhub_pass,
|
||||
text=True,
|
||||
check=True,
|
||||
cwd=REPO_ROOT,
|
||||
)
|
||||
else:
|
||||
log("DOCKERHUB_USERNAME/DOCKERHUB_TOKEN not set; assuming docker is already logged in")
|
||||
|
||||
tag = f"{dockerhub_repo}/{dockerhub_img}:{release_version}"
|
||||
log(f"Building and pushing {tag} via docker buildx bake")
|
||||
# docker-bake.hcl's DOCKERHUB_REPO/DOCKERHUB_IMG/RELEASE_VERSION are bake-level
|
||||
# `variable` blocks that compute `tags` - they read from env vars of the same
|
||||
# name, NOT from `--set target.args.*` (that sets Dockerfile build ARGs, a
|
||||
# separate namespace the Dockerfile doesn't even declare these under).
|
||||
bake_env = {
|
||||
**os.environ,
|
||||
"DOCKERHUB_REPO": dockerhub_repo,
|
||||
"DOCKERHUB_IMG": dockerhub_img,
|
||||
"RELEASE_VERSION": release_version,
|
||||
"HUGGINGFACE_ACCESS_TOKEN": hf_token,
|
||||
}
|
||||
subprocess.run(
|
||||
["docker", "buildx", "bake", "--push"],
|
||||
check=True,
|
||||
cwd=REPO_ROOT,
|
||||
env=bake_env,
|
||||
)
|
||||
return tag
|
||||
|
||||
|
||||
def api_request(method: str, url: str, api_key: str, body: dict | None = None) -> dict:
|
||||
data = json.dumps(body).encode() if body is not None else None
|
||||
req = urllib.request.Request(url, data=data, method=method)
|
||||
req.add_header("Authorization", f"Bearer {api_key}")
|
||||
req.add_header("Content-Type", "application/json")
|
||||
try:
|
||||
with urllib.request.urlopen(req) as resp:
|
||||
raw = resp.read()
|
||||
return json.loads(raw) if raw else {}
|
||||
except urllib.error.HTTPError as e:
|
||||
detail = e.read().decode(errors="replace")
|
||||
raise RuntimeError(f"{method} {url} -> HTTP {e.code}: {detail}") from e
|
||||
|
||||
|
||||
def create_template(api_key: str, name: str, image: str, env: dict, container_disk_in_gb: int) -> str:
|
||||
log(f"Creating template {name!r} for image {image}")
|
||||
resp = api_request("POST", f"{REST_API_BASE}/templates", api_key, {
|
||||
"name": name,
|
||||
"imageName": image,
|
||||
"isServerless": True,
|
||||
"env": env,
|
||||
"containerDiskInGb": container_disk_in_gb,
|
||||
})
|
||||
return resp["id"]
|
||||
|
||||
|
||||
def create_endpoint(
|
||||
api_key: str,
|
||||
name: str,
|
||||
template_id: str,
|
||||
gpu_type_ids: list[str],
|
||||
gpu_count: int,
|
||||
allowed_cuda_versions: list[str] | None,
|
||||
min_cuda_version: str | None,
|
||||
idle_timeout: int,
|
||||
) -> str:
|
||||
log(f"Creating endpoint {name!r} (gpuTypeIds={gpu_type_ids})")
|
||||
body = {
|
||||
"name": name,
|
||||
"templateId": template_id,
|
||||
"gpuTypeIds": gpu_type_ids,
|
||||
"gpuCount": gpu_count,
|
||||
"workersMin": 0,
|
||||
"workersMax": 1,
|
||||
"idleTimeout": idle_timeout,
|
||||
"scalerType": "QUEUE_DELAY",
|
||||
"scalerValue": 4,
|
||||
}
|
||||
if allowed_cuda_versions:
|
||||
body["allowedCudaVersions"] = allowed_cuda_versions
|
||||
if min_cuda_version:
|
||||
body["minCudaVersion"] = min_cuda_version
|
||||
resp = api_request("POST", f"{REST_API_BASE}/endpoints", api_key, body)
|
||||
return resp["id"]
|
||||
|
||||
|
||||
def delete_endpoint(api_key: str, endpoint_id: str) -> None:
|
||||
log(f"Deleting endpoint {endpoint_id}")
|
||||
try:
|
||||
api_request("DELETE", f"{REST_API_BASE}/endpoints/{endpoint_id}", api_key)
|
||||
except RuntimeError as e:
|
||||
log(f"WARNING: failed to delete endpoint {endpoint_id}: {e}")
|
||||
|
||||
|
||||
def delete_template(api_key: str, template_id: str) -> None:
|
||||
log(f"Deleting template {template_id}")
|
||||
try:
|
||||
api_request("DELETE", f"{REST_API_BASE}/templates/{template_id}", api_key)
|
||||
except RuntimeError as e:
|
||||
log(f"WARNING: failed to delete template {template_id}: {e}")
|
||||
|
||||
|
||||
def response_has_error(output) -> bool:
|
||||
if isinstance(output, dict):
|
||||
return "error" in output
|
||||
if isinstance(output, list):
|
||||
return any(isinstance(item, dict) and "error" in item for item in output)
|
||||
return False
|
||||
|
||||
|
||||
def run_test_case(api_key: str, endpoint_id: str, test: dict, cold_start_buffer_seconds: int) -> bool:
|
||||
name = test.get("name", "unnamed_test")
|
||||
deadline = time.monotonic() + test.get("timeout", 300000) / 1000 + cold_start_buffer_seconds
|
||||
|
||||
log(f"Submitting job for test {name!r}")
|
||||
submit = api_request("POST", f"{JOB_API_BASE}/{endpoint_id}/run", api_key, {"input": test["input"]})
|
||||
job_id = submit["id"]
|
||||
|
||||
while True:
|
||||
if time.monotonic() > deadline:
|
||||
log(f"FAIL {name}: timed out waiting for job {job_id}")
|
||||
return False
|
||||
|
||||
status_resp = api_request("GET", f"{JOB_API_BASE}/{endpoint_id}/status/{job_id}", api_key)
|
||||
status = status_resp.get("status")
|
||||
|
||||
if status in TERMINAL_STATUSES:
|
||||
if status != "COMPLETED":
|
||||
log(f"FAIL {name}: job {job_id} ended with status {status}: {status_resp}")
|
||||
return False
|
||||
if response_has_error(status_resp.get("output")):
|
||||
log(f"FAIL {name}: job {job_id} completed but output contained an error: {status_resp.get('output')}")
|
||||
return False
|
||||
log(f"PASS {name}")
|
||||
return True
|
||||
|
||||
time.sleep(5)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--config", required=True, type=Path, help="Path to a configs/**/*.yaml vLLM config")
|
||||
parser.add_argument("--image", help="Existing image tag to test; skips building unless --build is also given")
|
||||
parser.add_argument("--build", action="store_true", help="Build and push the image before testing")
|
||||
parser.add_argument("--keep", action="store_true", help="Don't tear down the endpoint/template afterward")
|
||||
parser.add_argument("--hub-json", type=Path, default=REPO_ROOT / ".runpod" / "hub.json")
|
||||
parser.add_argument("--tests-file", type=Path, default=REPO_ROOT / ".runpod" / "tests.json")
|
||||
parser.add_argument("--gpu-type-ids", default=",".join(DEFAULT_GPU_TYPE_IDS))
|
||||
parser.add_argument("--min-cuda-version", help="Overrides the config/hub.json minCudaVersion")
|
||||
parser.add_argument("--dockerhub-repo", default=os.environ.get("DOCKERHUB_REPO", "runpod"))
|
||||
parser.add_argument("--dockerhub-img", default=os.environ.get("DOCKERHUB_IMG", "worker-v1-vllm"))
|
||||
parser.add_argument("--idle-timeout", type=int, default=60)
|
||||
parser.add_argument("--cold-start-buffer-seconds", type=int, default=600)
|
||||
args = parser.parse_args()
|
||||
|
||||
api_key = os.environ.get("RUNPOD_API_KEY")
|
||||
if not api_key:
|
||||
log("ERROR: RUNPOD_API_KEY is not set")
|
||||
return 1
|
||||
|
||||
model_slug = args.config.stem
|
||||
run_id = uuid.uuid4().hex[:8]
|
||||
|
||||
if args.build or not args.image:
|
||||
release_version = f"test-{model_slug}-{run_id}"
|
||||
image = build_and_push_image(args.dockerhub_repo, args.dockerhub_img, release_version)
|
||||
else:
|
||||
image = args.image
|
||||
|
||||
hub_defaults = load_hub_defaults(args.hub_json)
|
||||
model_env = load_model_env(args.config)
|
||||
env = {**hub_defaults["env"], **model_env}
|
||||
hf_token = os.environ.get("HF_TOKEN") or os.environ.get("HUGGINGFACE_ACCESS_TOKEN")
|
||||
if hf_token:
|
||||
env["HF_TOKEN"] = hf_token
|
||||
tests_data = json.loads(args.tests_file.read_text())
|
||||
gpu_type_ids = [g.strip() for g in args.gpu_type_ids.split(",") if g.strip()]
|
||||
|
||||
resource_name = f"worker-vllm-e2e-{model_slug}-{run_id}"
|
||||
template_id = None
|
||||
endpoint_id = None
|
||||
try:
|
||||
template_id = create_template(
|
||||
api_key, resource_name, image, env, hub_defaults["containerDiskInGb"]
|
||||
)
|
||||
write_github_output("template_id", template_id)
|
||||
endpoint_id = create_endpoint(
|
||||
api_key,
|
||||
resource_name,
|
||||
template_id,
|
||||
gpu_type_ids,
|
||||
hub_defaults["gpuCount"],
|
||||
hub_defaults["allowedCudaVersions"],
|
||||
args.min_cuda_version or hub_defaults["minCudaVersion"],
|
||||
args.idle_timeout,
|
||||
)
|
||||
write_github_output("endpoint_id", endpoint_id)
|
||||
|
||||
results = [
|
||||
run_test_case(api_key, endpoint_id, test, args.cold_start_buffer_seconds)
|
||||
for test in tests_data["tests"]
|
||||
]
|
||||
|
||||
if all(results):
|
||||
log(f"All {len(results)} test(s) passed for {model_slug}")
|
||||
return 0
|
||||
log(f"{results.count(False)}/{len(results)} test(s) failed for {model_slug}")
|
||||
return 1
|
||||
finally:
|
||||
if not args.keep:
|
||||
if endpoint_id:
|
||||
delete_endpoint(api_key, endpoint_id)
|
||||
if template_id:
|
||||
delete_template(api_key, template_id)
|
||||
else:
|
||||
log(f"--keep passed; leaving endpoint={endpoint_id} template={template_id} running")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
+318
-53
@@ -1,43 +1,63 @@
|
||||
import os
|
||||
import logging
|
||||
import json
|
||||
import asyncio
|
||||
import inspect
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
from typing import AsyncGenerator, Optional
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from typing import AsyncGenerator, Optional
|
||||
import time
|
||||
|
||||
from vllm import AsyncLLMEngine
|
||||
from vllm.entrypoints.logger import RequestLogger
|
||||
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
|
||||
from vllm.entrypoints.openai.serving_completion import OpenAIServingCompletion
|
||||
from vllm.entrypoints.openai.protocol import ChatCompletionRequest, CompletionRequest, ErrorResponse
|
||||
from vllm.entrypoints.openai.serving_models import BaseModelPath, LoRAModulePath, OpenAIServingModels
|
||||
from vllm.inputs import TextPrompt
|
||||
from vllm.entrypoints.anthropic.protocol import AnthropicMessagesRequest, AnthropicMessagesResponse, AnthropicError, AnthropicErrorResponse
|
||||
from vllm.entrypoints.anthropic.serving import AnthropicServingMessages
|
||||
from vllm.entrypoints.openai.chat_completion.protocol import ChatCompletionRequest
|
||||
from vllm.entrypoints.openai.chat_completion.serving import OpenAIServingChat
|
||||
from vllm.entrypoints.openai.completion.protocol import CompletionRequest
|
||||
from vllm.entrypoints.openai.completion.serving import OpenAIServingCompletion
|
||||
from vllm.entrypoints.openai.engine.protocol import ErrorResponse
|
||||
from vllm.entrypoints.openai.models.protocol import BaseModelPath, LoRAModulePath
|
||||
from vllm.entrypoints.openai.models.serving import OpenAIServingModels
|
||||
from vllm.entrypoints.openai.responses.protocol import ResponsesRequest, ResponsesResponse
|
||||
from vllm.entrypoints.openai.responses.serving import OpenAIServingResponses
|
||||
from vllm.entrypoints.serve.render.serving import OpenAIServingRender
|
||||
|
||||
|
||||
from utils import DummyRequest, JobInput, BatchSize, create_error_response
|
||||
from constants import DEFAULT_MAX_CONCURRENCY, DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MIN_BATCH_SIZE
|
||||
from tokenizer import TokenizerWrapper
|
||||
from constants import DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MAX_CONCURRENCY, DEFAULT_MIN_BATCH_SIZE
|
||||
from engine_args import get_engine_args
|
||||
from tokenizer import TokenizerWrapper
|
||||
from utils import BatchSize, DummyRequest, JobInput, create_error_response
|
||||
|
||||
class vLLMEngine:
|
||||
def __init__(self, engine = None):
|
||||
load_dotenv() # For local development
|
||||
self.engine_args = get_engine_args()
|
||||
logging.info(f"Engine args: {self.engine_args}")
|
||||
|
||||
# Initialize vLLM engine first
|
||||
self.llm = self._initialize_llm() if engine is None else engine.llm
|
||||
if engine is None:
|
||||
ea = self.engine_args
|
||||
summary = {
|
||||
"model": ea.model,
|
||||
"dtype": ea.dtype,
|
||||
"quantization": ea.quantization,
|
||||
"max_model_len": ea.max_model_len,
|
||||
"tensor_parallel_size": ea.tensor_parallel_size,
|
||||
"gpu_memory_utilization": ea.gpu_memory_utilization,
|
||||
}
|
||||
if ea.tokenizer and ea.tokenizer != ea.model:
|
||||
summary["tokenizer"] = ea.tokenizer
|
||||
logging.info("Engine config: %s", summary)
|
||||
logging.debug("Full engine args: %s", ea)
|
||||
|
||||
# Only create custom tokenizer wrapper if not using mistral tokenizer mode
|
||||
# For mistral models, let vLLM handle tokenizer initialization
|
||||
if self.engine_args.tokenizer_mode != 'mistral':
|
||||
self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model,
|
||||
self.engine_args.tokenizer_revision,
|
||||
self.engine_args.trust_remote_code)
|
||||
self.llm = self._initialize_llm()
|
||||
|
||||
if self.engine_args.tokenizer_mode != 'mistral':
|
||||
self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model,
|
||||
self.engine_args.tokenizer_revision,
|
||||
self.engine_args.trust_remote_code)
|
||||
else:
|
||||
self.tokenizer = None
|
||||
else:
|
||||
# For mistral models, we'll get the tokenizer from vLLM later
|
||||
self.tokenizer = None
|
||||
self.llm = engine.llm
|
||||
self.tokenizer = engine.tokenizer
|
||||
|
||||
self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
|
||||
self.default_batch_size = int(os.getenv("DEFAULT_BATCH_SIZE", DEFAULT_BATCH_SIZE))
|
||||
@@ -110,7 +130,7 @@ class vLLMEngine:
|
||||
if apply_chat_template or isinstance(llm_input, list):
|
||||
tokenizer_wrapper = self._get_tokenizer_for_chat_template()
|
||||
llm_input = tokenizer_wrapper.apply_chat_template(llm_input)
|
||||
results_generator = self.llm.generate(llm_input, validated_sampling_params, request_id)
|
||||
results_generator = self.llm.generate(TextPrompt(prompt=llm_input), validated_sampling_params, request_id)
|
||||
n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True
|
||||
last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0}
|
||||
|
||||
@@ -174,10 +194,24 @@ class vLLMEngine:
|
||||
class OpenAIvLLMEngine(vLLMEngine):
|
||||
def __init__(self, vllm_engine):
|
||||
super().__init__(vllm_engine)
|
||||
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.engine_args.model
|
||||
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.engine_args.served_model_name or self.engine_args.model
|
||||
self.response_role = os.getenv("OPENAI_RESPONSE_ROLE") or "assistant"
|
||||
self.lora_adapters = self._load_lora_adapters()
|
||||
asyncio.run(self._initialize_engines())
|
||||
|
||||
# Always defer OpenAI engine initialization to the first request.
|
||||
# asyncio.run() creates a temporary event loop that gets closed, but async
|
||||
# components (tokenizer pool, serving engines) bind futures to that loop.
|
||||
# When Runpod's serverless handler runs in its own event loop, those futures
|
||||
# are "attached to a different loop" causing RuntimeError.
|
||||
# This affects all configurations, not just LoRA.
|
||||
self._engines_initialized = False
|
||||
if self.lora_adapters:
|
||||
logging.info(f"LoRA mode: {len(self.lora_adapters)} adapter(s) will load on first request")
|
||||
for adapter in self.lora_adapters:
|
||||
logging.info(f" - {adapter.name}: {adapter.path}")
|
||||
else:
|
||||
logging.info("OpenAI engines will initialize on first request")
|
||||
|
||||
# Handle both integer and boolean string values for RAW_OPENAI_OUTPUT
|
||||
raw_output_env = os.getenv("RAW_OPENAI_OUTPUT", "1")
|
||||
if raw_output_env.lower() in ('true', 'false'):
|
||||
@@ -186,30 +220,72 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
self.raw_openai_output = bool(int(raw_output_env))
|
||||
|
||||
def _load_lora_adapters(self):
|
||||
adapters = []
|
||||
try:
|
||||
adapters = json.loads(os.getenv("LORA_MODULES", '[]'))
|
||||
except Exception as e:
|
||||
logging.info(f"---Initialized adapter json load error: {e}")
|
||||
lora_modules_env = os.getenv("LORA_MODULES", "")
|
||||
if not lora_modules_env:
|
||||
return []
|
||||
|
||||
for i, adapter in enumerate(adapters):
|
||||
try:
|
||||
parsed = json.loads(lora_modules_env)
|
||||
except json.JSONDecodeError as e:
|
||||
logging.error(
|
||||
"LORA_MODULES could not be parsed as JSON: %s — no LoRA adapters loaded. Value: %r",
|
||||
e, lora_modules_env,
|
||||
)
|
||||
return []
|
||||
|
||||
# Accept a single adapter dict as well as an array
|
||||
if isinstance(parsed, dict):
|
||||
parsed = [parsed]
|
||||
|
||||
if not isinstance(parsed, list):
|
||||
logging.error(
|
||||
"LORA_MODULES must be a JSON array of adapter objects, got %s — no LoRA adapters loaded.",
|
||||
type(parsed).__name__,
|
||||
)
|
||||
return []
|
||||
|
||||
adapters = []
|
||||
for i, adapter in enumerate(parsed):
|
||||
try:
|
||||
adapters[i] = LoRAModulePath(**adapter)
|
||||
logging.info(f"---Initialized adapter: {adapter}")
|
||||
adapters.append(LoRAModulePath(**adapter))
|
||||
logging.info("Loaded LoRA adapter config [%d]: %s", i, adapter)
|
||||
except Exception as e:
|
||||
logging.info(f"---Initialized adapter not worked: {e}")
|
||||
continue
|
||||
logging.error(
|
||||
"Failed to parse LoRA adapter at index %d: %s. Config: %r",
|
||||
i, e, adapter,
|
||||
)
|
||||
|
||||
if parsed and not adapters:
|
||||
logging.error(
|
||||
"LORA_MODULES specified %d adapter(s) but none could be loaded — "
|
||||
"OpenAI model name lookups for LoRA adapters will fail.",
|
||||
len(parsed),
|
||||
)
|
||||
|
||||
return adapters
|
||||
|
||||
async def _ensure_engines_initialized(self):
|
||||
"""Initialize engines on first request to avoid event loop mismatch.
|
||||
|
||||
In Runpod Serverless, the startup code runs outside the handler's event
|
||||
loop. Deferring initialization to the first request ensures all async
|
||||
components (tokenizer pool, serving engines, LoRA state) are created in
|
||||
the correct event loop context.
|
||||
"""
|
||||
if not self._engines_initialized:
|
||||
logging.info("Initializing OpenAI serving engines...")
|
||||
await self._initialize_engines()
|
||||
self._engines_initialized = True
|
||||
logging.info("OpenAI serving engines initialized successfully")
|
||||
|
||||
async def _initialize_engines(self):
|
||||
self.model_config = await self.llm.get_model_config()
|
||||
self.model_config = self.llm.model_config
|
||||
self.base_model_paths = [
|
||||
BaseModelPath(name=self.engine_args.model, model_path=self.engine_args.model)
|
||||
BaseModelPath(name=self.served_model_name, model_path=self.engine_args.model)
|
||||
]
|
||||
|
||||
self.serving_models = OpenAIServingModels(
|
||||
engine_client=self.llm,
|
||||
model_config=self.model_config,
|
||||
base_model_paths=self.base_model_paths,
|
||||
lora_modules=self.lora_adapters,
|
||||
)
|
||||
@@ -220,35 +296,101 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
if self.tokenizer and hasattr(self.tokenizer, 'tokenizer'):
|
||||
chat_template = self.tokenizer.tokenizer.chat_template
|
||||
|
||||
self.chat_engine = OpenAIServingChat(
|
||||
engine_client=self.llm,
|
||||
model_config=self.model_config,
|
||||
models=self.serving_models,
|
||||
response_role=self.response_role,
|
||||
self.openai_serving_render = OpenAIServingRender(
|
||||
model_config=self.llm.model_config,
|
||||
renderer=self.llm.renderer,
|
||||
model_registry=self.serving_models.registry,
|
||||
request_logger=None,
|
||||
chat_template=chat_template,
|
||||
chat_template_content_format="auto",
|
||||
# enable_reasoning=os.getenv('ENABLE_REASONING', 'false').lower() == 'true',
|
||||
reasoning_parser= os.getenv('REASONING_PARSER', "") or None,
|
||||
# return_token_as_token_ids=False,
|
||||
trust_request_chat_template=os.getenv('TRUST_REQUEST_CHAT_TEMPLATE', 'false').lower() == 'true',
|
||||
enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true',
|
||||
exclude_tools_when_tool_choice_none=os.getenv('EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE', 'false').lower() == 'true',
|
||||
tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None,
|
||||
enable_prompt_tokens_details=False
|
||||
reasoning_parser=os.getenv('REASONING_PARSER', "") or None,
|
||||
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
|
||||
)
|
||||
|
||||
self.chat_engine = OpenAIServingChat(
|
||||
engine_client=self.llm,
|
||||
models=self.serving_models,
|
||||
response_role=self.response_role,
|
||||
openai_serving_render=self.openai_serving_render,
|
||||
request_logger=None,
|
||||
chat_template=chat_template,
|
||||
chat_template_content_format="auto",
|
||||
trust_request_chat_template=os.getenv('TRUST_REQUEST_CHAT_TEMPLATE', 'false').lower() == 'true',
|
||||
return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true',
|
||||
reasoning_parser=os.getenv('REASONING_PARSER', "") or "",
|
||||
enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true',
|
||||
exclude_tools_when_tool_choice_none=os.getenv('EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE', 'false').lower() == 'true',
|
||||
tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None,
|
||||
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
|
||||
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
||||
enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true',
|
||||
)
|
||||
self.completion_engine = OpenAIServingCompletion(
|
||||
engine_client=self.llm,
|
||||
model_config=self.model_config,
|
||||
models=self.serving_models,
|
||||
openai_serving_render=self.openai_serving_render,
|
||||
request_logger=None,
|
||||
# return_token_as_token_ids=False,
|
||||
return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true',
|
||||
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
|
||||
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
||||
)
|
||||
self.responses_engine = OpenAIServingResponses(
|
||||
engine_client=self.llm,
|
||||
models=self.serving_models,
|
||||
openai_serving_render=self.openai_serving_render,
|
||||
request_logger=None,
|
||||
chat_template=chat_template,
|
||||
chat_template_content_format="auto",
|
||||
return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true',
|
||||
reasoning_parser=os.getenv('REASONING_PARSER', "") or "",
|
||||
enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true',
|
||||
tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None,
|
||||
tool_server=None,
|
||||
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
|
||||
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
||||
enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true',
|
||||
)
|
||||
self.messages_engine = AnthropicServingMessages(
|
||||
engine_client=self.llm,
|
||||
models=self.serving_models,
|
||||
response_role=self.response_role,
|
||||
openai_serving_render=self.openai_serving_render,
|
||||
request_logger=None,
|
||||
chat_template=chat_template,
|
||||
chat_template_content_format="auto",
|
||||
return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true',
|
||||
reasoning_parser=os.getenv('REASONING_PARSER', "") or "",
|
||||
enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true',
|
||||
tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None,
|
||||
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
|
||||
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
||||
)
|
||||
|
||||
warmup = getattr(self.chat_engine, 'warmup', None)
|
||||
if callable(warmup):
|
||||
result = warmup()
|
||||
if inspect.isawaitable(result):
|
||||
await result
|
||||
|
||||
async def generate(self, openai_request: JobInput):
|
||||
# Ensure engines are ready (no-op if already initialized at startup)
|
||||
await self._ensure_engines_initialized()
|
||||
|
||||
if openai_request.openai_route == "/v1/models":
|
||||
yield await self._handle_model_request()
|
||||
elif openai_request.openai_route in ["/v1/chat/completions", "/v1/completions"]:
|
||||
async for response in self._handle_chat_or_completion_request(openai_request):
|
||||
yield response
|
||||
elif openai_request.openai_route == "/v1/responses":
|
||||
async for response in self._handle_responses_request(openai_request):
|
||||
yield response
|
||||
elif openai_request.openai_route == "/v1/messages":
|
||||
async for response in self._handle_messages_request(openai_request):
|
||||
yield response
|
||||
else:
|
||||
yield create_error_response("Invalid route").model_dump()
|
||||
|
||||
@@ -304,3 +446,126 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
batch = "".join(batch)
|
||||
yield batch
|
||||
|
||||
async def _handle_responses_request(self, openai_request: JobInput):
|
||||
request_id = getattr(openai_request, "request_id", "unknown")
|
||||
|
||||
try:
|
||||
request = ResponsesRequest(**openai_request.openai_input)
|
||||
except Exception as e:
|
||||
logging.error(
|
||||
"Invalid ResponsesRequest JSON: %s",
|
||||
e,
|
||||
extra={"request_id": request_id}
|
||||
)
|
||||
yield create_error_response(
|
||||
"Invalid request format",
|
||||
err_type="BadRequestError"
|
||||
).model_dump()
|
||||
return
|
||||
|
||||
dummy_request = DummyRequest()
|
||||
try:
|
||||
response = await self.responses_engine.create_responses(request, raw_request=dummy_request)
|
||||
except Exception as e:
|
||||
logging.error(
|
||||
"Failed to create Responses: %s",
|
||||
e,
|
||||
extra={"request_id": request_id},
|
||||
exc_info=True
|
||||
)
|
||||
yield create_error_response(
|
||||
"Internal server error during response generation",
|
||||
err_type="InternalServerError"
|
||||
).model_dump()
|
||||
return
|
||||
|
||||
if isinstance(response, (ErrorResponse, ResponsesResponse)):
|
||||
yield response.model_dump()
|
||||
return
|
||||
|
||||
try:
|
||||
async for event in response:
|
||||
if not hasattr(event, "type"):
|
||||
continue
|
||||
event_type = getattr(event, "type", "unknown")
|
||||
yield f"event: {event_type}\ndata: {event.model_dump_json(indent=None)}\n\n"
|
||||
except Exception as e:
|
||||
logging.error(
|
||||
"Error processing responses stream: %s",
|
||||
e,
|
||||
extra={"request_id": request_id},
|
||||
exc_info=True
|
||||
)
|
||||
error_payload = create_error_response(
|
||||
"Streaming response failed",
|
||||
err_type="InternalServerError"
|
||||
).model_dump_json()
|
||||
yield f"event: error\ndata: {error_payload}\n\n"
|
||||
|
||||
async def _handle_messages_request(self, openai_request: JobInput):
|
||||
request_id = getattr(openai_request, "request_id", "unknown")
|
||||
|
||||
try:
|
||||
request = AnthropicMessagesRequest(**openai_request.openai_input)
|
||||
except Exception as e:
|
||||
logging.error(
|
||||
"Invalid AnthropicMessagesRequest: %s",
|
||||
e,
|
||||
extra={"request_id": request_id}
|
||||
)
|
||||
yield AnthropicErrorResponse(
|
||||
error=AnthropicError(
|
||||
type="invalid_request_error",
|
||||
message="Invalid request format"
|
||||
)
|
||||
).model_dump()
|
||||
return
|
||||
|
||||
dummy_request = DummyRequest()
|
||||
|
||||
try:
|
||||
response = await self.messages_engine.create_messages(request, raw_request=dummy_request)
|
||||
except Exception as e:
|
||||
logging.error(
|
||||
"Failed to create messages: %s",
|
||||
e,
|
||||
extra={"request_id": request_id},
|
||||
exc_info=True
|
||||
)
|
||||
yield AnthropicErrorResponse(
|
||||
error=AnthropicError(
|
||||
type="internal_error",
|
||||
message="Failed to generate messages"
|
||||
)
|
||||
).model_dump()
|
||||
return
|
||||
|
||||
if isinstance(response, ErrorResponse):
|
||||
error_type = getattr(response, "type", "internal_error")
|
||||
error_message = getattr(response, "message", "Unknown error")
|
||||
yield AnthropicErrorResponse(
|
||||
error=AnthropicError(type=error_type, message=error_message)
|
||||
).model_dump()
|
||||
return
|
||||
|
||||
if isinstance(response, AnthropicMessagesResponse):
|
||||
yield response.model_dump(exclude_none=True)
|
||||
return
|
||||
|
||||
try:
|
||||
async for chunk in response:
|
||||
yield chunk
|
||||
except Exception as e:
|
||||
logging.error(
|
||||
"Error streaming messages: %s",
|
||||
e,
|
||||
extra={"request_id": request_id},
|
||||
exc_info=True
|
||||
)
|
||||
error_payload = AnthropicErrorResponse(
|
||||
error=AnthropicError(
|
||||
type="internal_error",
|
||||
message="Error while streaming messages"
|
||||
)
|
||||
).model_dump_json()
|
||||
yield f"event: error\ndata: {error_payload}\n\n"
|
||||
|
||||
+539
-105
@@ -1,118 +1,426 @@
|
||||
import ast
|
||||
import os
|
||||
import json
|
||||
import logging
|
||||
from typing import get_origin, get_args
|
||||
from torch.cuda import device_count
|
||||
from vllm import AsyncEngineArgs
|
||||
from vllm.model_executor.model_loader.tensorizer import TensorizerConfig
|
||||
from src.utils import convert_limit_mm_per_prompt
|
||||
|
||||
RENAME_ARGS_MAP = {
|
||||
# Backward-compat: env var names users already know → engine arg name
|
||||
ENV_ALIASES = {
|
||||
"MODEL_NAME": "model",
|
||||
"MODEL_REVISION": "revision",
|
||||
"TOKENIZER_NAME": "tokenizer",
|
||||
"MAX_CONTEXT_LEN_TO_CAPTURE": "max_seq_len_to_capture"
|
||||
}
|
||||
|
||||
# Literal defaults from original worker (used when env/local do not set a value)
|
||||
DEFAULT_ARGS = {
|
||||
"disable_log_stats": os.getenv('DISABLE_LOG_STATS', 'False').lower() == 'true',
|
||||
"disable_log_requests": os.getenv('DISABLE_LOG_REQUESTS', 'False').lower() == 'true',
|
||||
"gpu_memory_utilization": float(os.getenv('GPU_MEMORY_UTILIZATION', 0.95)),
|
||||
"pipeline_parallel_size": int(os.getenv('PIPELINE_PARALLEL_SIZE', 1)),
|
||||
"tensor_parallel_size": int(os.getenv('TENSOR_PARALLEL_SIZE', 1)),
|
||||
"served_model_name": os.getenv('SERVED_MODEL_NAME', None),
|
||||
"tokenizer": os.getenv('TOKENIZER', None),
|
||||
"skip_tokenizer_init": os.getenv('SKIP_TOKENIZER_INIT', 'False').lower() == 'true',
|
||||
"tokenizer_mode": os.getenv('TOKENIZER_MODE', 'auto'),
|
||||
"trust_remote_code": os.getenv('TRUST_REMOTE_CODE', 'False').lower() == 'true',
|
||||
"download_dir": os.getenv('DOWNLOAD_DIR', None),
|
||||
"load_format": os.getenv('LOAD_FORMAT', 'auto'),
|
||||
"config_format": os.getenv('CONFIG_FORMAT', 'auto'),
|
||||
"dtype": os.getenv('DTYPE', 'auto'),
|
||||
"kv_cache_dtype": os.getenv('KV_CACHE_DTYPE', 'auto'),
|
||||
"quantization_param_path": os.getenv('QUANTIZATION_PARAM_PATH', None),
|
||||
"seed": int(os.getenv('SEED', 0)),
|
||||
"max_model_len": int(os.getenv('MAX_MODEL_LEN', 0)) or None,
|
||||
"worker_use_ray": os.getenv('WORKER_USE_RAY', 'False').lower() == 'true',
|
||||
"distributed_executor_backend": os.getenv('DISTRIBUTED_EXECUTOR_BACKEND', None),
|
||||
"max_parallel_loading_workers": int(os.getenv('MAX_PARALLEL_LOADING_WORKERS', 0)) or None,
|
||||
"block_size": int(os.getenv('BLOCK_SIZE', 16)),
|
||||
"enable_prefix_caching": os.getenv('ENABLE_PREFIX_CACHING', 'False').lower() == 'true',
|
||||
"disable_sliding_window": os.getenv('DISABLE_SLIDING_WINDOW', 'False').lower() == 'true',
|
||||
"use_v2_block_manager": os.getenv('USE_V2_BLOCK_MANAGER', 'False').lower() == 'true',
|
||||
"swap_space": int(os.getenv('SWAP_SPACE', 4)), # GiB
|
||||
"cpu_offload_gb": int(os.getenv('CPU_OFFLOAD_GB', 0)), # GiB
|
||||
"max_num_batched_tokens": int(os.getenv('MAX_NUM_BATCHED_TOKENS', 0)) or None,
|
||||
"max_num_seqs": int(os.getenv('MAX_NUM_SEQS', 256)),
|
||||
"max_logprobs": int(os.getenv('MAX_LOGPROBS', 20)), # Default value for OpenAI Chat Completions API
|
||||
"revision": os.getenv('REVISION', None),
|
||||
"code_revision": os.getenv('CODE_REVISION', None),
|
||||
"rope_scaling": os.getenv('ROPE_SCALING', None),
|
||||
"rope_theta": float(os.getenv('ROPE_THETA', 0)) or None,
|
||||
"tokenizer_revision": os.getenv('TOKENIZER_REVISION', None),
|
||||
"quantization": os.getenv('QUANTIZATION', None),
|
||||
"enforce_eager": os.getenv('ENFORCE_EAGER', 'False').lower() == 'true',
|
||||
"max_context_len_to_capture": int(os.getenv('MAX_CONTEXT_LEN_TO_CAPTURE', 0)) or None,
|
||||
"max_seq_len_to_capture": int(os.getenv('MAX_SEQ_LEN_TO_CAPTURE', 8192)),
|
||||
"disable_custom_all_reduce": os.getenv('DISABLE_CUSTOM_ALL_REDUCE', 'False').lower() == 'true',
|
||||
"tokenizer_pool_size": int(os.getenv('TOKENIZER_POOL_SIZE', 0)),
|
||||
"tokenizer_pool_type": os.getenv('TOKENIZER_POOL_TYPE', 'ray'),
|
||||
"tokenizer_pool_extra_config": os.getenv('TOKENIZER_POOL_EXTRA_CONFIG', None),
|
||||
"enable_lora": os.getenv('ENABLE_LORA', 'False').lower() == 'true',
|
||||
"max_loras": int(os.getenv('MAX_LORAS', 1)),
|
||||
"max_lora_rank": int(os.getenv('MAX_LORA_RANK', 16)),
|
||||
"enable_prompt_adapter": os.getenv('ENABLE_PROMPT_ADAPTER', 'False').lower() == 'true',
|
||||
"max_prompt_adapters": int(os.getenv('MAX_PROMPT_ADAPTERS', 1)),
|
||||
"max_prompt_adapter_token": int(os.getenv('MAX_PROMPT_ADAPTER_TOKEN', 0)),
|
||||
"fully_sharded_loras": os.getenv('FULLY_SHARDED_LORAS', 'False').lower() == 'true',
|
||||
"lora_extra_vocab_size": int(os.getenv('LORA_EXTRA_VOCAB_SIZE', 256)),
|
||||
"long_lora_scaling_factors": tuple(map(float, os.getenv('LONG_LORA_SCALING_FACTORS', '').split(','))) if os.getenv('LONG_LORA_SCALING_FACTORS') else None,
|
||||
"lora_dtype": os.getenv('LORA_DTYPE', 'auto'),
|
||||
"max_cpu_loras": int(os.getenv('MAX_CPU_LORAS', 0)) or None,
|
||||
"device": os.getenv('DEVICE', 'auto'),
|
||||
"ray_workers_use_nsight": os.getenv('RAY_WORKERS_USE_NSIGHT', 'False').lower() == 'true',
|
||||
"num_gpu_blocks_override": int(os.getenv('NUM_GPU_BLOCKS_OVERRIDE', 0)) or None,
|
||||
"num_lookahead_slots": int(os.getenv('NUM_LOOKAHEAD_SLOTS', 0)),
|
||||
"model_loader_extra_config": os.getenv('MODEL_LOADER_EXTRA_CONFIG', None),
|
||||
"ignore_patterns": os.getenv('IGNORE_PATTERNS', None),
|
||||
"preemption_mode": os.getenv('PREEMPTION_MODE', None),
|
||||
"scheduler_delay_factor": float(os.getenv('SCHEDULER_DELAY_FACTOR', 0.0)),
|
||||
"enable_chunked_prefill": os.getenv('ENABLE_CHUNKED_PREFILL', None),
|
||||
"guided_decoding_backend": os.getenv('GUIDED_DECODING_BACKEND', 'outlines'),
|
||||
"speculative_model": os.getenv('SPECULATIVE_MODEL', None),
|
||||
"speculative_draft_tensor_parallel_size": int(os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE', 0)) or None,
|
||||
"enable_expert_parallel": bool(os.getenv('ENABLE_EXPERT_PARALLEL', 'False').lower() == 'true'),
|
||||
"num_speculative_tokens": int(os.getenv('NUM_SPECULATIVE_TOKENS', 0)) or None,
|
||||
"speculative_max_model_len": int(os.getenv('SPECULATIVE_MAX_MODEL_LEN', 0)) or None,
|
||||
"speculative_disable_by_batch_size": int(os.getenv('SPECULATIVE_DISABLE_BY_BATCH_SIZE', 0)) or None,
|
||||
"ngram_prompt_lookup_max": int(os.getenv('NGRAM_PROMPT_LOOKUP_MAX', 0)) or None,
|
||||
"ngram_prompt_lookup_min": int(os.getenv('NGRAM_PROMPT_LOOKUP_MIN', 0)) or None,
|
||||
"spec_decoding_acceptance_method": os.getenv('SPEC_DECODING_ACCEPTANCE_METHOD', 'rejection_sampler'),
|
||||
"typical_acceptance_sampler_posterior_threshold": float(os.getenv('TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD', 0)) or None,
|
||||
"typical_acceptance_sampler_posterior_alpha": float(os.getenv('TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA', 0)) or None,
|
||||
"qlora_adapter_name_or_path": os.getenv('QLORA_ADAPTER_NAME_OR_PATH', None),
|
||||
"disable_logprobs_during_spec_decoding": os.getenv('DISABLE_LOGPROBS_DURING_SPEC_DECODING', None),
|
||||
"otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None),
|
||||
"use_v2_block_manager": os.getenv('USE_V2_BLOCK_MANAGER', 'true'),
|
||||
"disable_log_stats": False,
|
||||
"enable_log_requests": False,
|
||||
"gpu_memory_utilization": 0.95,
|
||||
"pipeline_parallel_size": 1,
|
||||
"tensor_parallel_size": 1,
|
||||
"skip_tokenizer_init": False,
|
||||
"tokenizer_mode": "auto",
|
||||
"trust_remote_code": False,
|
||||
"load_format": "auto",
|
||||
"dtype": "auto",
|
||||
"kv_cache_dtype": "auto",
|
||||
"seed": 0,
|
||||
"worker_use_ray": False,
|
||||
"block_size": 16,
|
||||
"enable_prefix_caching": False,
|
||||
"disable_sliding_window": False,
|
||||
"swap_space": 4,
|
||||
"cpu_offload_gb": 0,
|
||||
"max_num_seqs": 256,
|
||||
"max_logprobs": 20,
|
||||
"enforce_eager": False,
|
||||
"max_seq_len_to_capture": 8192,
|
||||
"disable_custom_all_reduce": False,
|
||||
"tokenizer_pool_size": 0,
|
||||
"tokenizer_pool_type": "ray",
|
||||
"enable_lora": False,
|
||||
"max_loras": 1,
|
||||
"max_lora_rank": 16,
|
||||
"enable_prompt_adapter": False,
|
||||
"max_prompt_adapters": 1,
|
||||
"max_prompt_adapter_token": 0,
|
||||
"fully_sharded_loras": False,
|
||||
"lora_extra_vocab_size": 256,
|
||||
"lora_dtype": "auto",
|
||||
"device": "auto",
|
||||
"ray_workers_use_nsight": False,
|
||||
"num_lookahead_slots": 0,
|
||||
"scheduler_delay_factor": 0.0,
|
||||
"guided_decoding_backend": "outlines",
|
||||
"spec_decoding_acceptance_method": "rejection_sampler",
|
||||
"stream_interval": 1,
|
||||
|
||||
}
|
||||
limit_mm_env = os.getenv('LIMIT_MM_PER_PROMPT')
|
||||
if limit_mm_env is not None:
|
||||
DEFAULT_ARGS["limit_mm_per_prompt"] = convert_limit_mm_per_prompt(limit_mm_env)
|
||||
|
||||
def match_vllm_args(args):
|
||||
"""Rename args to match vllm by:
|
||||
1. Renaming keys to lower case
|
||||
2. Renaming keys to match vllm
|
||||
3. Filtering args to match vllm's AsyncEngineArgs
|
||||
|
||||
Args:
|
||||
args (dict): Dictionary of args
|
||||
def _resolve_field_type(field_type: type) -> type:
|
||||
"""Resolve Optional/Union to the concrete type for conversion."""
|
||||
origin = get_origin(field_type)
|
||||
args = get_args(field_type) if hasattr(field_type, "__args__") else ()
|
||||
if origin is not None:
|
||||
# Optional[X] is Union[X, None]; X | None is UnionType
|
||||
non_none = [a for a in args if a is not type(None)]
|
||||
if non_none:
|
||||
return non_none[0]
|
||||
return field_type
|
||||
|
||||
Returns:
|
||||
dict: Dictionary of args with renamed keys
|
||||
|
||||
def _convert_env_value_to_field_type(value: str, field_name: str, field_type: type):
|
||||
"""Convert env var string to the type expected by AsyncEngineArgs for this field."""
|
||||
val = value.strip() if isinstance(value, str) else value
|
||||
if val in ("", "None", "none"):
|
||||
args = get_args(field_type) if hasattr(field_type, "__args__") else ()
|
||||
if type(None) in (args or ()):
|
||||
return None
|
||||
raise ValueError("empty value not allowed for non-optional field")
|
||||
|
||||
# Union[bool, str, ...]: only coerce to bool for unambiguous literals;
|
||||
# otherwise preserve the string (e.g. hf_token="hf_abc..." must stay a str).
|
||||
if get_origin(field_type) is not None:
|
||||
union_types = [a for a in (get_args(field_type) or ()) if a is not type(None)]
|
||||
if bool in union_types and str in union_types:
|
||||
if str(val).lower() in ("true", "false", "1", "0", "yes", "no", "on", "off"):
|
||||
return str(val).lower() in ("true", "1", "yes", "on")
|
||||
return str(val)
|
||||
|
||||
effective_type = _resolve_field_type(field_type)
|
||||
# bool
|
||||
if effective_type is bool:
|
||||
return str(val).lower() in ("true", "1", "yes", "on")
|
||||
# int
|
||||
if effective_type is int:
|
||||
return int(val)
|
||||
# float
|
||||
if effective_type is float:
|
||||
return float(val)
|
||||
# str
|
||||
if effective_type is str:
|
||||
return str(val)
|
||||
# dict, list, or complex (try JSON)
|
||||
origin = get_origin(effective_type)
|
||||
if effective_type in (dict, list) or origin in (dict, list):
|
||||
try:
|
||||
return json.loads(val)
|
||||
except json.JSONDecodeError:
|
||||
return val
|
||||
# tuple (e.g. long_lora_scaling_factors) — comma-separated or JSON array
|
||||
if effective_type is tuple or origin is tuple:
|
||||
args = get_args(field_type) if hasattr(field_type, "__args__") else ()
|
||||
elem_types = [a for a in args if a is not Ellipsis]
|
||||
elem_type = elem_types[0] if elem_types else str
|
||||
try:
|
||||
parsed = json.loads(val)
|
||||
if isinstance(parsed, list):
|
||||
return tuple(elem_type(x) for x in parsed)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
pass
|
||||
return tuple(elem_type(x.strip()) for x in str(val).split(",") if x.strip())
|
||||
# For dataclass/complex types, try JSON then Python literal parsing to dict
|
||||
try:
|
||||
return json.loads(val)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
pass
|
||||
try:
|
||||
parsed = ast.literal_eval(val)
|
||||
if isinstance(parsed, (dict, list)):
|
||||
return parsed
|
||||
except (ValueError, SyntaxError):
|
||||
pass
|
||||
# Fallback: try int, float, then str
|
||||
try:
|
||||
return int(val)
|
||||
except ValueError:
|
||||
pass
|
||||
try:
|
||||
return float(val)
|
||||
except ValueError:
|
||||
pass
|
||||
return str(val)
|
||||
|
||||
|
||||
def _get_args_from_env_auto_discover() -> dict:
|
||||
"""Auto-discover engine args from env vars using UPPERCASED field names.
|
||||
|
||||
For every field in AsyncEngineArgs, check os.getenv(FIELD_NAME).
|
||||
E.g. MAX_MODEL_LEN=4096 -> max_model_len=4096.
|
||||
Uses same type conversion as before; supports all vLLM engine args without manual listing.
|
||||
"""
|
||||
renamed_args = {RENAME_ARGS_MAP.get(k, k): v for k, v in args.items()}
|
||||
matched_args = {k: v for k, v in renamed_args.items() if k in AsyncEngineArgs.__dataclass_fields__}
|
||||
return {k: v for k, v in matched_args.items() if v not in [None, "", "None"]}
|
||||
args = {}
|
||||
valid_fields = AsyncEngineArgs.__dataclass_fields__
|
||||
for field_name, field in valid_fields.items():
|
||||
env_key = field_name.upper()
|
||||
value = os.environ.get(env_key)
|
||||
if value is None:
|
||||
continue
|
||||
try:
|
||||
args[field_name] = _convert_env_value_to_field_type(
|
||||
value, field_name, field.type
|
||||
)
|
||||
except (ValueError, TypeError, json.JSONDecodeError) as e:
|
||||
logging.warning(
|
||||
"Skip env %s=%r: %s", env_key, value, e
|
||||
)
|
||||
return args
|
||||
|
||||
|
||||
def _apply_env_aliases(args: dict) -> None:
|
||||
"""Apply ENV_ALIASES: if MODEL_NAME etc. are set, set the target engine arg."""
|
||||
valid_fields = AsyncEngineArgs.__dataclass_fields__
|
||||
for alias, target in ENV_ALIASES.items():
|
||||
value = os.environ.get(alias)
|
||||
if value is None or target not in valid_fields:
|
||||
continue
|
||||
try:
|
||||
args[target] = _convert_env_value_to_field_type(
|
||||
value, target, valid_fields[target].type
|
||||
)
|
||||
except (ValueError, TypeError, json.JSONDecodeError) as e:
|
||||
logging.warning("Skip env alias %s=%r: %s", alias, value, e)
|
||||
|
||||
def get_speculative_config():
|
||||
"""Build speculative decoding configuration from environment variables.
|
||||
|
||||
Supports two modes:
|
||||
1. Full JSON config via SPECULATIVE_CONFIG env var
|
||||
2. Individual env vars for common settings
|
||||
"""
|
||||
# Option 1: Full JSON configuration
|
||||
spec_config_json = os.getenv('SPECULATIVE_CONFIG')
|
||||
if spec_config_json:
|
||||
try:
|
||||
config = json.loads(spec_config_json)
|
||||
logging.info(f"Using speculative config from SPECULATIVE_CONFIG: {config}")
|
||||
return config
|
||||
except json.JSONDecodeError as e:
|
||||
logging.error(f"Failed to parse SPECULATIVE_CONFIG JSON: {e}")
|
||||
return None
|
||||
|
||||
# Option 2: Build config from individual environment variables
|
||||
spec_method = os.getenv('SPECULATIVE_METHOD')
|
||||
spec_model = os.getenv('SPECULATIVE_MODEL')
|
||||
_num_spec_tokens = os.getenv('NUM_SPECULATIVE_TOKENS')
|
||||
_ngram_max = os.getenv('NGRAM_PROMPT_LOOKUP_MAX')
|
||||
_ngram_min = os.getenv('NGRAM_PROMPT_LOOKUP_MIN')
|
||||
|
||||
# Convert numeric vars to int so '0' (hub.json default) is treated as unset
|
||||
num_spec_tokens = (int(_num_spec_tokens) or None) if _num_spec_tokens else None
|
||||
ngram_max = (int(_ngram_max) or None) if _ngram_max else None
|
||||
ngram_min = (int(_ngram_min) or None) if _ngram_min else None
|
||||
|
||||
if not any([spec_method, spec_model, ngram_max]):
|
||||
return None
|
||||
|
||||
config = {}
|
||||
|
||||
# Determine method
|
||||
if spec_method:
|
||||
config['method'] = spec_method
|
||||
elif ngram_max and not spec_model:
|
||||
config['method'] = 'ngram'
|
||||
elif spec_model:
|
||||
model_lower = spec_model.lower()
|
||||
if 'eagle3' in model_lower:
|
||||
config['method'] = 'eagle3'
|
||||
elif 'eagle' in model_lower:
|
||||
config['method'] = 'eagle'
|
||||
elif 'medusa' in model_lower:
|
||||
config['method'] = 'medusa'
|
||||
else:
|
||||
config['method'] = 'draft_model'
|
||||
|
||||
if spec_model:
|
||||
config['model'] = spec_model
|
||||
if num_spec_tokens:
|
||||
config['num_speculative_tokens'] = num_spec_tokens
|
||||
if ngram_max:
|
||||
config['prompt_lookup_max'] = ngram_max
|
||||
if ngram_min:
|
||||
config['prompt_lookup_min'] = ngram_min
|
||||
|
||||
draft_tp = os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE')
|
||||
if draft_tp:
|
||||
config['draft_tensor_parallel_size'] = int(draft_tp)
|
||||
|
||||
spec_max_len = os.getenv('SPECULATIVE_MAX_MODEL_LEN')
|
||||
if spec_max_len:
|
||||
config['max_model_len'] = int(spec_max_len)
|
||||
|
||||
disable_batch = os.getenv('SPECULATIVE_DISABLE_BY_BATCH_SIZE')
|
||||
if disable_batch:
|
||||
config['disable_by_batch_size'] = int(disable_batch)
|
||||
|
||||
spec_quant = os.getenv('SPECULATIVE_QUANTIZATION')
|
||||
if spec_quant:
|
||||
config['quantization'] = spec_quant
|
||||
|
||||
spec_revision = os.getenv('SPECULATIVE_MODEL_REVISION')
|
||||
if spec_revision:
|
||||
config['revision'] = spec_revision
|
||||
|
||||
spec_eager = os.getenv('SPECULATIVE_ENFORCE_EAGER')
|
||||
if spec_eager:
|
||||
config['enforce_eager'] = spec_eager.lower() == 'true'
|
||||
|
||||
if config:
|
||||
logging.info(f"Built speculative config from env vars: {config}")
|
||||
return config
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def _resolve_max_model_len(model, trust_remote_code=False, revision=None):
|
||||
"""Resolve max_model_len from the model's HuggingFace config."""
|
||||
try:
|
||||
from transformers import AutoConfig
|
||||
config = AutoConfig.from_pretrained(
|
||||
model,
|
||||
trust_remote_code=trust_remote_code,
|
||||
revision=revision,
|
||||
)
|
||||
for attr in ('max_position_embeddings', 'n_positions', 'max_seq_len', 'seq_length'):
|
||||
val = getattr(config, attr, None)
|
||||
if val is not None:
|
||||
logging.info(f"Resolved max_model_len={val} from model config ({attr})")
|
||||
return val
|
||||
except Exception as e:
|
||||
logging.warning(f"Could not resolve max_model_len from model config: {e}")
|
||||
return None
|
||||
|
||||
|
||||
def _local_args_to_engine_args(local: dict) -> dict:
|
||||
"""Map local args (e.g. from /local_model_args.json) to engine arg names and filter."""
|
||||
valid = AsyncEngineArgs.__dataclass_fields__
|
||||
out = {}
|
||||
for k, v in local.items():
|
||||
target = ENV_ALIASES.get(k, k.lower().replace("-", "_"))
|
||||
if target not in valid or v in (None, "", "None"):
|
||||
continue
|
||||
out[target] = v
|
||||
return out
|
||||
|
||||
|
||||
def _sanitize_hf_overrides(hf_overrides: dict) -> dict | None:
|
||||
"""Strip rope_scaling from hf_overrides sub-configs if vLLM rejects them.
|
||||
|
||||
Older vLLM (<0.7) required explicit mrope rope_scaling in hf_overrides for
|
||||
models like Qwen2-VL. Newer vLLM auto-detects mrope and raises a ValueError
|
||||
in patch_rope_scaling_dict when it finds conflicting rope_type values. Strip
|
||||
the offending rope_scaling so the model loads with its native config.
|
||||
"""
|
||||
if not isinstance(hf_overrides, dict):
|
||||
return hf_overrides
|
||||
|
||||
try:
|
||||
from vllm.transformers_utils.config import patch_rope_scaling_dict
|
||||
except ImportError:
|
||||
return hf_overrides
|
||||
|
||||
import copy
|
||||
cleaned = {}
|
||||
changed = False
|
||||
for key, value in hf_overrides.items():
|
||||
if isinstance(value, dict) and "rope_scaling" in value:
|
||||
rope_scaling = value.get("rope_scaling")
|
||||
if isinstance(rope_scaling, dict):
|
||||
try:
|
||||
patch_rope_scaling_dict(copy.deepcopy(rope_scaling))
|
||||
except (ValueError, Exception) as e:
|
||||
logging.warning(
|
||||
"Stripping hf_overrides['%s']['rope_scaling'] because vLLM "
|
||||
"rejected it (%s). Newer vLLM auto-detects rope scaling from "
|
||||
"the model config.", key, e
|
||||
)
|
||||
stripped = {k: v for k, v in value.items() if k != "rope_scaling"}
|
||||
cleaned[key] = stripped if stripped else None
|
||||
changed = True
|
||||
continue
|
||||
cleaned[key] = value
|
||||
|
||||
if not changed:
|
||||
return hf_overrides
|
||||
|
||||
result = {k: v for k, v in cleaned.items() if v is not None}
|
||||
return result or None
|
||||
|
||||
|
||||
def _resolve_cached_model_path(model_name: str) -> str:
|
||||
"""Return a local snapshot path when the HF cache was stored with lowercase names.
|
||||
|
||||
Some model stores (e.g. RunPod pre-cached volumes) normalize repo IDs to
|
||||
lowercase. HuggingFace Hub stores caches as
|
||||
``models--{org}--{model}/snapshots/{hash}/`` preserving the original casing,
|
||||
so MODEL_NAME=Qwen/Qwen2.5-Coder-32B-Instruct-AWQ will miss a cache stored
|
||||
as ``models--qwen--qwen2.5-coder-32b-instruct-awq/``.
|
||||
|
||||
If the exact-case cache directory is absent but a lowercase variant exists,
|
||||
the latest snapshot path is returned so vLLM loads from disk rather than
|
||||
attempting a redundant download.
|
||||
"""
|
||||
if os.path.isabs(model_name):
|
||||
return model_name
|
||||
|
||||
cache_dir = (
|
||||
os.getenv("HUGGINGFACE_HUB_CACHE")
|
||||
or os.getenv("HF_HOME")
|
||||
or os.path.expanduser("~/.cache/huggingface/hub")
|
||||
)
|
||||
|
||||
folder_name = f"models--{model_name.replace('/', '--')}"
|
||||
|
||||
if os.path.isdir(os.path.join(cache_dir, folder_name)):
|
||||
return model_name
|
||||
|
||||
lower_dir = os.path.join(cache_dir, folder_name.lower())
|
||||
if not os.path.isdir(lower_dir):
|
||||
return model_name
|
||||
|
||||
snapshots_dir = os.path.join(lower_dir, "snapshots")
|
||||
if not os.path.isdir(snapshots_dir):
|
||||
return model_name
|
||||
|
||||
try:
|
||||
snapshots = sorted(os.listdir(snapshots_dir))
|
||||
except OSError:
|
||||
return model_name
|
||||
|
||||
if not snapshots:
|
||||
return model_name
|
||||
|
||||
resolved = os.path.join(snapshots_dir, snapshots[-1])
|
||||
logging.info(
|
||||
"MODEL_NAME %r not found at original casing in HF cache; "
|
||||
"resolved to lowercase cached snapshot at %r",
|
||||
model_name, resolved,
|
||||
)
|
||||
return resolved
|
||||
|
||||
|
||||
def _get_args_from_config_file() -> dict:
|
||||
"""Load engine args from a vLLM-style config.yaml.
|
||||
|
||||
Checks VLLM_CONFIG_FILE env var, then falls back to /vllm_config.yaml.
|
||||
Keys use the same long-form names as vllm serve (hyphens converted to underscores).
|
||||
"""
|
||||
import yaml
|
||||
path = os.getenv("VLLM_CONFIG_FILE", "/vllm_config.yaml")
|
||||
if not os.path.exists(path):
|
||||
return {}
|
||||
with open(path) as f:
|
||||
raw = yaml.safe_load(f) or {}
|
||||
normalized = {k.replace("-", "_"): v for k, v in raw.items()}
|
||||
logging.info("Loaded engine args from config file %s: %s", path, list(normalized.keys()))
|
||||
return normalized
|
||||
|
||||
|
||||
def get_local_args():
|
||||
"""
|
||||
Retrieve local arguments from a JSON file.
|
||||
@@ -135,23 +443,46 @@ def get_local_args():
|
||||
|
||||
return local_args
|
||||
def get_engine_args():
|
||||
# Start with default args
|
||||
args = DEFAULT_ARGS
|
||||
# Start with worker custom defaults (only where we differ from vLLM)
|
||||
args = dict(DEFAULT_ARGS)
|
||||
|
||||
# Get env args that match keys in AsyncEngineArgs
|
||||
args.update(os.environ)
|
||||
# Config file values sit above defaults but below env vars
|
||||
args.update(_get_args_from_config_file())
|
||||
|
||||
# Get local args if model is baked in and overwrite env args
|
||||
args.update(get_local_args())
|
||||
# Auto-discover: every AsyncEngineArgs field from env UPPERCASED (e.g. MAX_MODEL_LEN)
|
||||
args.update(_get_args_from_env_auto_discover())
|
||||
|
||||
# Backward-compat aliases (MODEL_NAME → model, etc.)
|
||||
_apply_env_aliases(args)
|
||||
|
||||
# Local baked-in model overrides
|
||||
local = get_local_args()
|
||||
if local:
|
||||
args.update(_local_args_to_engine_args(local))
|
||||
|
||||
# Filter to valid engine args and drop sentinel empty values
|
||||
valid_fields = AsyncEngineArgs.__dataclass_fields__
|
||||
args = {
|
||||
k: v for k, v in args.items()
|
||||
if k in valid_fields and v not in (None, "", "None")
|
||||
}
|
||||
|
||||
# Special conversion for limit_mm_per_prompt (e.g. "image=1,video=0")
|
||||
limit_mm_env = os.getenv("LIMIT_MM_PER_PROMPT")
|
||||
if limit_mm_env is not None:
|
||||
args["limit_mm_per_prompt"] = convert_limit_mm_per_prompt(limit_mm_env)
|
||||
|
||||
# if args.get("TENSORIZER_URI"): TODO: add back once tensorizer is ready
|
||||
# args["load_format"] = "tensorizer"
|
||||
# args["model_loader_extra_config"] = TensorizerConfig(tensorizer_uri=args["TENSORIZER_URI"], num_readers=None)
|
||||
# logging.info(f"Using tensorized model from {args['TENSORIZER_URI']}")
|
||||
|
||||
|
||||
# Rename and match to vllm args
|
||||
args = match_vllm_args(args)
|
||||
if "hf_overrides" in args:
|
||||
sanitized = _sanitize_hf_overrides(args["hf_overrides"])
|
||||
if sanitized:
|
||||
args["hf_overrides"] = sanitized
|
||||
else:
|
||||
del args["hf_overrides"]
|
||||
|
||||
if args.get("load_format") == "bitsandbytes":
|
||||
args["quantization"] = args["load_format"]
|
||||
@@ -164,6 +495,53 @@ def get_engine_args():
|
||||
if os.getenv("MAX_PARALLEL_LOADING_WORKERS"):
|
||||
logging.warning("Overriding MAX_PARALLEL_LOADING_WORKERS with None because more than 1 GPU is available.")
|
||||
|
||||
# LMCache requires HMA to be disabled
|
||||
try:
|
||||
_kv_transfer = args.get("kv_transfer_config")
|
||||
if isinstance(_kv_transfer, str):
|
||||
parsed = None
|
||||
try:
|
||||
parsed = json.loads(_kv_transfer)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
pass
|
||||
if parsed is None:
|
||||
try:
|
||||
result = ast.literal_eval(_kv_transfer)
|
||||
if isinstance(result, dict):
|
||||
parsed = result
|
||||
except (ValueError, SyntaxError):
|
||||
pass
|
||||
if parsed is not None:
|
||||
_kv_transfer = parsed
|
||||
args["kv_transfer_config"] = _kv_transfer
|
||||
_kv_offload = args.get("kv_offloading_backend")
|
||||
|
||||
lmcache_via_offload = _kv_offload == "lmcache"
|
||||
lmcache_via_transfer = (
|
||||
isinstance(_kv_transfer, dict)
|
||||
and isinstance(_kv_transfer.get("kv_connector"), str)
|
||||
and "lmcache" in _kv_transfer.get("kv_connector", "").lower()
|
||||
)
|
||||
lmcache_detected = lmcache_via_offload or lmcache_via_transfer
|
||||
|
||||
if lmcache_detected:
|
||||
current = args.get("disable_hybrid_kv_cache_manager")
|
||||
if current is False:
|
||||
logging.warning(
|
||||
"disable_hybrid_kv_cache_manager=False conflicts with LMCache; "
|
||||
"overriding to True (HMA must be disabled when using LMCache)"
|
||||
)
|
||||
args["disable_hybrid_kv_cache_manager"] = True
|
||||
elif current is None:
|
||||
args["disable_hybrid_kv_cache_manager"] = True
|
||||
logging.info("LMCache detected: automatically setting disable_hybrid_kv_cache_manager=True")
|
||||
except Exception as e:
|
||||
logging.error(
|
||||
"Failed to check LMCache configuration: %s",
|
||||
e,
|
||||
exc_info=True
|
||||
)
|
||||
|
||||
# Deprecated env args backwards compatibility
|
||||
if args.get("kv_cache_dtype") == "fp8_e5m2":
|
||||
args["kv_cache_dtype"] = "fp8"
|
||||
@@ -176,4 +554,60 @@ def get_engine_args():
|
||||
# os.environ["VLLM_ATTENTION_BACKEND"] = "FLASHINFER"
|
||||
# logging.info("Using FLASHINFER for gemma-2 model.")
|
||||
|
||||
# Set max_num_batched_tokens to max_model_len for unlimited batching.
|
||||
# vLLM defaults max_num_batched_tokens to 2048 when None, which is too low.
|
||||
|
||||
if args.get("max_model_len") == 0:
|
||||
args["max_model_len"] = None
|
||||
|
||||
if args.get("max_num_batched_tokens") == 0:
|
||||
args["max_num_batched_tokens"] = None
|
||||
|
||||
if args.get("max_num_batched_tokens") is None:
|
||||
max_model_len = args.get("max_model_len")
|
||||
if max_model_len is None:
|
||||
max_model_len = _resolve_max_model_len(
|
||||
args.get("model"),
|
||||
trust_remote_code=args.get("trust_remote_code", False),
|
||||
revision=args.get("revision"),
|
||||
)
|
||||
if max_model_len is not None:
|
||||
args["max_num_batched_tokens"] = max_model_len
|
||||
logging.info(f"Setting max_num_batched_tokens to {max_model_len}")
|
||||
|
||||
# VLLM_ATTENTION_BACKEND is deprecated, migrate to attention_backend
|
||||
if os.getenv('VLLM_ATTENTION_BACKEND'):
|
||||
logging.warning(
|
||||
"VLLM_ATTENTION_BACKEND env var is deprecated. "
|
||||
"Use ATTENTION_BACKEND instead (maps to --attention-backend CLI arg)."
|
||||
)
|
||||
if not args.get('attention_backend'):
|
||||
args['attention_backend'] = os.getenv('VLLM_ATTENTION_BACKEND')
|
||||
|
||||
# DISABLE_LOG_REQUESTS is deprecated, use ENABLE_LOG_REQUESTS instead
|
||||
if os.getenv('DISABLE_LOG_REQUESTS'):
|
||||
logging.warning(
|
||||
"DISABLE_LOG_REQUESTS env var is deprecated. "
|
||||
"Use ENABLE_LOG_REQUESTS instead (default: False)."
|
||||
)
|
||||
# Honor old behavior: if DISABLE_LOG_REQUESTS=true, don't enable logging
|
||||
if os.getenv('DISABLE_LOG_REQUESTS', 'False').lower() == 'true':
|
||||
args['enable_log_requests'] = False
|
||||
|
||||
# Add speculative decoding configuration if present
|
||||
speculative_config = get_speculative_config()
|
||||
if speculative_config:
|
||||
args["speculative_config"] = speculative_config
|
||||
|
||||
# Resolve lowercase HF cache paths (FDE-174)
|
||||
if args.get("model"):
|
||||
original_model = args["model"]
|
||||
args["model"] = _resolve_cached_model_path(original_model)
|
||||
# When the model was rewritten to an on-disk snapshot path, keep serving
|
||||
# under the original repo id so the OpenAI API model name does not become
|
||||
# a filesystem path (issue #310). An explicit served_model_name (or the
|
||||
# OPENAI_SERVED_MODEL_NAME_OVERRIDE handled downstream) still wins.
|
||||
if args["model"] != original_model and not args.get("served_model_name"):
|
||||
args["served_model_name"] = original_model
|
||||
|
||||
return AsyncEngineArgs(**args)
|
||||
|
||||
+50
-17
@@ -1,22 +1,55 @@
|
||||
import os
|
||||
import sys
|
||||
import multiprocessing
|
||||
import traceback
|
||||
import runpod
|
||||
from utils import JobInput
|
||||
from engine import vLLMEngine, OpenAIvLLMEngine
|
||||
from runpod import RunPodLogger
|
||||
|
||||
log = RunPodLogger()
|
||||
|
||||
vllm_engine = None
|
||||
openai_engine = None
|
||||
|
||||
vllm_engine = vLLMEngine()
|
||||
OpenAIvLLMEngine = OpenAIvLLMEngine(vllm_engine)
|
||||
|
||||
async def handler(job):
|
||||
job_input = JobInput(job["input"])
|
||||
engine = OpenAIvLLMEngine if job_input.openai_route else vllm_engine
|
||||
results_generator = engine.generate(job_input)
|
||||
async for batch in results_generator:
|
||||
yield batch
|
||||
try:
|
||||
from utils import JobInput
|
||||
job_input = JobInput(job["input"])
|
||||
engine = openai_engine if job_input.openai_route else vllm_engine
|
||||
results_generator = engine.generate(job_input)
|
||||
async for batch in results_generator:
|
||||
yield batch
|
||||
except Exception as e:
|
||||
error_str = str(e)
|
||||
full_traceback = traceback.format_exc()
|
||||
|
||||
runpod.serverless.start(
|
||||
{
|
||||
"handler": handler,
|
||||
"concurrency_modifier": lambda x: vllm_engine.max_concurrency,
|
||||
"return_aggregate_stream": True,
|
||||
}
|
||||
)
|
||||
log.error(f"Error during inference: {error_str}")
|
||||
log.error(f"Full traceback:\n{full_traceback}")
|
||||
|
||||
# CUDA errors = worker is broken, exit to let RunPod spin up a healthy one
|
||||
if "CUDA" in error_str or "cuda" in error_str:
|
||||
log.error("Terminating worker due to CUDA/GPU error")
|
||||
sys.exit(1)
|
||||
|
||||
yield {"error": error_str}
|
||||
|
||||
|
||||
# Only run in main process to prevent re-initialization when vLLM spawns worker subprocesses
|
||||
if __name__ == "__main__" or multiprocessing.current_process().name == "MainProcess":
|
||||
|
||||
try:
|
||||
from engine import vLLMEngine, OpenAIvLLMEngine
|
||||
|
||||
vllm_engine = vLLMEngine()
|
||||
openai_engine = OpenAIvLLMEngine(vllm_engine)
|
||||
log.info("vLLM engines initialized successfully")
|
||||
except Exception as e:
|
||||
log.error(f"Worker startup failed: {e}\n{traceback.format_exc()}")
|
||||
sys.exit(1)
|
||||
|
||||
runpod.serverless.start(
|
||||
{
|
||||
"handler": handler,
|
||||
"concurrency_modifier": lambda x: vllm_engine.max_concurrency if vllm_engine else 1,
|
||||
"return_aggregate_stream": True,
|
||||
}
|
||||
)
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
#!/bin/bash
|
||||
set -e
|
||||
|
||||
if [ -n "${TRANSFORMERS_VERSION}" ]; then
|
||||
echo "Installing transformers==${TRANSFORMERS_VERSION}"
|
||||
uv pip install --system "transformers==${TRANSFORMERS_VERSION}"
|
||||
fi
|
||||
|
||||
exec python3 /src/handler.py
|
||||
+4
-2
@@ -1,10 +1,12 @@
|
||||
from transformers import AutoTokenizer
|
||||
import logging
|
||||
import os
|
||||
from typing import Union
|
||||
|
||||
from transformers import AutoTokenizer
|
||||
|
||||
class TokenizerWrapper:
|
||||
def __init__(self, tokenizer_name_or_path, tokenizer_revision, trust_remote_code):
|
||||
print(f"tokenizer_name_or_path: {tokenizer_name_or_path}, tokenizer_revision: {tokenizer_revision}, trust_remote_code: {trust_remote_code}")
|
||||
logging.debug("tokenizer_name_or_path: %s, tokenizer_revision: %s, trust_remote_code: %s", tokenizer_name_or_path, tokenizer_revision, trust_remote_code)
|
||||
self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_name_or_path, revision=tokenizer_revision or "main", trust_remote_code=trust_remote_code)
|
||||
self.custom_chat_template = os.getenv("CUSTOM_CHAT_TEMPLATE")
|
||||
self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template)
|
||||
|
||||
+4
-5
@@ -3,11 +3,10 @@ import logging
|
||||
from http import HTTPStatus
|
||||
from functools import wraps
|
||||
from time import time
|
||||
from vllm.entrypoints.openai.protocol import RequestResponseMetadata
|
||||
|
||||
try:
|
||||
from vllm.utils import random_uuid
|
||||
from vllm.entrypoints.openai.protocol import ErrorResponse
|
||||
from vllm.entrypoints.openai.engine.protocol import ErrorResponse, ErrorInfo, RequestResponseMetadata
|
||||
from vllm import SamplingParams
|
||||
except ImportError:
|
||||
logging.warning("Error importing vllm, skipping related imports. This is ONLY expected when baking model into docker image from a machine without GPUs")
|
||||
@@ -88,9 +87,9 @@ class BatchSize:
|
||||
self.current_batch_size = min(self.current_batch_size*self.batch_size_growth_factor, self.max_batch_size)
|
||||
|
||||
def create_error_response(message: str, err_type: str = "BadRequestError", status_code: HTTPStatus = HTTPStatus.BAD_REQUEST) -> ErrorResponse:
|
||||
return ErrorResponse(message=message,
|
||||
type=err_type,
|
||||
code=status_code.value)
|
||||
return ErrorResponse(error=ErrorInfo(message=message,
|
||||
type=err_type,
|
||||
code=status_code.value))
|
||||
|
||||
def get_int_bool_env(env_var: str, default: bool) -> bool:
|
||||
return int(os.getenv(env_var, int(default))) == 1
|
||||
|
||||
@@ -0,0 +1,105 @@
|
||||
"""Shared test fixtures.
|
||||
|
||||
``src/engine_args.py`` hard-imports ``vllm`` (and a tensorizer submodule) and
|
||||
``torch.cuda``. Both are only installed inside the GPU Docker image, so when the
|
||||
tests run on a machine without them we install lightweight stubs. When the real
|
||||
packages *are* available (e.g. CI inside the worker image) the stubs are skipped
|
||||
and the real ones are used instead.
|
||||
"""
|
||||
|
||||
import sys
|
||||
import types
|
||||
from dataclasses import dataclass
|
||||
from typing import Optional, Union, List
|
||||
|
||||
|
||||
def _install_torch_stub():
|
||||
try:
|
||||
import torch # noqa: F401
|
||||
return # real torch present, nothing to stub
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
torch = types.ModuleType("torch")
|
||||
cuda = types.ModuleType("torch.cuda")
|
||||
# No GPU in the test environment -> 0 devices (skips tensor-parallel setup).
|
||||
cuda.device_count = lambda: 0
|
||||
torch.cuda = cuda
|
||||
sys.modules["torch"] = torch
|
||||
sys.modules["torch.cuda"] = cuda
|
||||
|
||||
|
||||
def _install_vllm_stub():
|
||||
try:
|
||||
import vllm # noqa: F401
|
||||
return # real vLLM present, nothing to stub
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
vllm = types.ModuleType("vllm")
|
||||
|
||||
@dataclass
|
||||
class AsyncEngineArgs:
|
||||
# Only the fields the worker actually sets/reads need to exist here;
|
||||
# get_engine_args() filters args down to AsyncEngineArgs.__dataclass_fields__
|
||||
# before construction, so unknown keys are dropped rather than passed.
|
||||
model: Optional[str] = None
|
||||
served_model_name: Optional[Union[str, List[str]]] = None
|
||||
revision: Optional[str] = None
|
||||
tokenizer: Optional[str] = None
|
||||
trust_remote_code: bool = False
|
||||
max_model_len: Optional[int] = None
|
||||
max_num_batched_tokens: Optional[int] = None
|
||||
disable_log_stats: bool = False
|
||||
gpu_memory_utilization: float = 0.9
|
||||
tensor_parallel_size: int = 1
|
||||
max_parallel_loading_workers: Optional[int] = None
|
||||
kv_cache_dtype: Optional[str] = None
|
||||
|
||||
class _Stub: # pragma: no cover - placeholder for vllm symbols
|
||||
def __init__(self, *args, **kwargs):
|
||||
pass
|
||||
|
||||
vllm.AsyncEngineArgs = AsyncEngineArgs
|
||||
vllm.SamplingParams = _Stub
|
||||
sys.modules["vllm"] = vllm
|
||||
|
||||
# src.utils imports these at module load and uses ErrorResponse as a return
|
||||
# annotation, which Python evaluates eagerly on <3.14 -> must be defined.
|
||||
vllm_utils = types.ModuleType("vllm.utils")
|
||||
vllm_utils.random_uuid = lambda: "stub-uuid"
|
||||
vllm.utils = vllm_utils
|
||||
sys.modules["vllm.utils"] = vllm_utils
|
||||
|
||||
protocol = types.ModuleType("vllm.entrypoints.openai.engine.protocol")
|
||||
protocol.ErrorResponse = _Stub
|
||||
protocol.ErrorInfo = _Stub
|
||||
protocol.RequestResponseMetadata = _Stub
|
||||
for name in (
|
||||
"vllm.entrypoints",
|
||||
"vllm.entrypoints.openai",
|
||||
"vllm.entrypoints.openai.engine",
|
||||
):
|
||||
sys.modules.setdefault(name, types.ModuleType(name))
|
||||
sys.modules["vllm.entrypoints.openai.engine.protocol"] = protocol
|
||||
|
||||
# vllm.model_executor.model_loader.tensorizer.TensorizerConfig
|
||||
model_executor = types.ModuleType("vllm.model_executor")
|
||||
model_loader = types.ModuleType("vllm.model_executor.model_loader")
|
||||
tensorizer = types.ModuleType("vllm.model_executor.model_loader.tensorizer")
|
||||
|
||||
class TensorizerConfig: # pragma: no cover - placeholder
|
||||
def __init__(self, *args, **kwargs):
|
||||
pass
|
||||
|
||||
tensorizer.TensorizerConfig = TensorizerConfig
|
||||
model_loader.tensorizer = tensorizer
|
||||
model_executor.model_loader = model_loader
|
||||
vllm.model_executor = model_executor
|
||||
sys.modules["vllm.model_executor"] = model_executor
|
||||
sys.modules["vllm.model_executor.model_loader"] = model_loader
|
||||
sys.modules["vllm.model_executor.model_loader.tensorizer"] = tensorizer
|
||||
|
||||
|
||||
_install_torch_stub()
|
||||
_install_vllm_stub()
|
||||
@@ -0,0 +1,6 @@
|
||||
# Test-only dependencies. vllm/torch are stubbed in conftest.py when absent,
|
||||
# so the unit tests run on a plain CPU runner without the GPU image.
|
||||
pytest>=8,<10
|
||||
# get_engine_args() reads a vLLM-style config via PyYAML (a transitive vllm dep
|
||||
# at runtime); install it explicitly here since vllm itself is stubbed.
|
||||
pyyaml
|
||||
@@ -0,0 +1,126 @@
|
||||
"""Tests for HF cache path resolution and served-model-name decoupling.
|
||||
|
||||
Regression coverage for issue #310: when MODEL_NAME is served from a lowercased
|
||||
HF cache dir, the cache resolver rewrites engine_args.model to a snapshot path.
|
||||
The served model name must stay the original repo id, not the path.
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
import pytest
|
||||
|
||||
from src import engine_args
|
||||
from src.engine_args import _resolve_cached_model_path, get_engine_args
|
||||
|
||||
|
||||
MODEL = "Qwen/Qwen3.6-27B-FP8"
|
||||
SNAPSHOT_HASH = "e89b16ebf1988b3d6befa7de50abc2d76f26eb09"
|
||||
|
||||
|
||||
def _make_cache(root, folder_name, snapshot=SNAPSHOT_HASH):
|
||||
"""Create a HF-style ``models--…/snapshots/<hash>/`` dir and return its path."""
|
||||
snap_dir = os.path.join(root, folder_name, "snapshots", snapshot)
|
||||
os.makedirs(snap_dir)
|
||||
return snap_dir
|
||||
|
||||
|
||||
def _is_case_sensitive_fs(path):
|
||||
"""The lowercase-cache resolution only matters on case-sensitive filesystems.
|
||||
|
||||
On macOS (APFS, case-insensitive by default) ``models--Qwen--…`` and
|
||||
``models--qwen--…`` collide, so the resolver always sees the exact-case dir
|
||||
as present. Production runs on Linux (case-sensitive), which is what these
|
||||
tests exercise.
|
||||
"""
|
||||
probe = os.path.join(path, "CaseProbe")
|
||||
open(probe, "w").close()
|
||||
try:
|
||||
return not os.path.exists(os.path.join(path, "caseprobe"))
|
||||
finally:
|
||||
os.remove(probe)
|
||||
|
||||
|
||||
requires_case_sensitive_fs = pytest.mark.skipif(
|
||||
not _is_case_sensitive_fs(os.environ.get("TMPDIR", "/tmp")),
|
||||
reason="lowercase HF cache resolution only applies on case-sensitive filesystems",
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def hf_cache(tmp_path, monkeypatch):
|
||||
cache = tmp_path / "hub"
|
||||
cache.mkdir()
|
||||
monkeypatch.setenv("HUGGINGFACE_HUB_CACHE", str(cache))
|
||||
# Make sure HF_HOME does not shadow the explicit cache dir during the test.
|
||||
monkeypatch.delenv("HF_HOME", raising=False)
|
||||
return cache
|
||||
|
||||
|
||||
class TestResolveCachedModelPath:
|
||||
def test_exact_case_dir_returns_repo_id(self, hf_cache):
|
||||
_make_cache(str(hf_cache), "models--Qwen--Qwen3.6-27B-FP8")
|
||||
assert _resolve_cached_model_path(MODEL) == MODEL
|
||||
|
||||
def test_no_cache_returns_repo_id(self, hf_cache):
|
||||
assert _resolve_cached_model_path(MODEL) == MODEL
|
||||
|
||||
def test_absolute_path_passthrough(self, hf_cache):
|
||||
path = "/runpod-volume/some/local/model"
|
||||
assert _resolve_cached_model_path(path) == path
|
||||
|
||||
@requires_case_sensitive_fs
|
||||
def test_lowercase_dir_returns_snapshot_path(self, hf_cache):
|
||||
snap = _make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")
|
||||
assert _resolve_cached_model_path(MODEL) == snap
|
||||
|
||||
def test_lowercase_dir_without_snapshots_returns_repo_id(self, hf_cache):
|
||||
# Dir exists but has no snapshots subdir -> nothing to resolve to.
|
||||
os.makedirs(os.path.join(str(hf_cache), "models--qwen--qwen3.6-27b-fp8"))
|
||||
assert _resolve_cached_model_path(MODEL) == MODEL
|
||||
|
||||
@requires_case_sensitive_fs
|
||||
def test_lowercase_dir_picks_latest_snapshot(self, hf_cache):
|
||||
folder = "models--qwen--qwen3.6-27b-fp8"
|
||||
_make_cache(str(hf_cache), folder, snapshot="aaaa")
|
||||
latest = _make_cache(str(hf_cache), folder, snapshot="zzzz")
|
||||
assert _resolve_cached_model_path(MODEL) == latest
|
||||
|
||||
|
||||
class TestGetEngineArgsServedName:
|
||||
"""Issue #310: served name must be decoupled from the resolved on-disk path."""
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def base_env(self, monkeypatch):
|
||||
# Avoid the network branch in _resolve_max_model_len.
|
||||
monkeypatch.setenv("MAX_NUM_BATCHED_TOKENS", "2048")
|
||||
monkeypatch.delenv("SERVED_MODEL_NAME", raising=False)
|
||||
# Don't pick up a stray vLLM config file from the environment.
|
||||
monkeypatch.setenv("VLLM_CONFIG_FILE", "/nonexistent-vllm-config.yaml")
|
||||
|
||||
@requires_case_sensitive_fs
|
||||
def test_served_name_is_repo_id_when_path_rewritten(self, hf_cache, monkeypatch):
|
||||
snap = _make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")
|
||||
monkeypatch.setenv("MODEL_NAME", MODEL)
|
||||
|
||||
result = get_engine_args()
|
||||
|
||||
assert result.model == snap # weights load from the lowercase cache
|
||||
assert result.served_model_name == MODEL # API still serves the repo id
|
||||
|
||||
def test_served_name_untouched_when_no_rewrite(self, hf_cache, monkeypatch):
|
||||
_make_cache(str(hf_cache), "models--Qwen--Qwen3.6-27B-FP8")
|
||||
monkeypatch.setenv("MODEL_NAME", MODEL)
|
||||
|
||||
result = get_engine_args()
|
||||
|
||||
assert result.model == MODEL
|
||||
assert result.served_model_name is None
|
||||
|
||||
def test_explicit_served_name_not_overridden(self, hf_cache, monkeypatch):
|
||||
_make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")
|
||||
monkeypatch.setenv("MODEL_NAME", MODEL)
|
||||
monkeypatch.setenv("SERVED_MODEL_NAME", "custom-name")
|
||||
|
||||
result = get_engine_args()
|
||||
|
||||
assert result.served_model_name == "custom-name"
|
||||
Reference in New Issue
Block a user