Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7351da512b | ||
|
|
1e78043b2c | ||
|
|
352c64f4c1 | ||
|
|
0922f5b435 | ||
|
|
5d9a48fc70 | ||
|
|
5d1579e361 | ||
|
|
0488b77d89 | ||
|
|
d3a962c33b | ||
|
|
105c125698 | ||
|
|
0a0ccfcb60 | ||
|
|
c8ce53c72c | ||
|
|
9618e799ba | ||
|
|
cb3f077dba | ||
|
|
69646b9e99 | ||
|
|
8b991a7ad7 | ||
|
|
dac05b62b3 | ||
|
|
d356c31675 | ||
|
|
14b74a4989 | ||
|
|
80072047ab | ||
|
|
50aba8fb57 | ||
|
|
9edc5715ce | ||
|
|
4c91f2c5b5 | ||
|
|
6265b99348 | ||
|
|
026f8d700b | ||
|
|
146bdb0252 | ||
|
|
da01193a3d | ||
|
|
c2e6cc9f61 | ||
|
|
69968a6b39 | ||
|
|
32b29d4c6c | ||
|
|
dcea4fc4f9 | ||
|
|
9c139e8ceb | ||
|
|
678bb4be8f | ||
|
|
87d7365126 | ||
|
|
0e83616f93 | ||
|
|
ab6d39dcf8 | ||
|
|
ed315a175e | ||
|
|
73f030ae5e | ||
|
|
8a099c1723 | ||
|
|
ff87840a58 | ||
|
|
7dc853b1fe | ||
|
|
a8b754b92a | ||
|
|
6357aeda51 | ||
|
|
0140b29c44 | ||
|
|
0cb8aeae77 | ||
|
|
cd8f9e9560 | ||
|
|
895fd25fac | ||
|
|
7bb8df73af | ||
|
|
747cdf5891 | ||
|
|
3d4af5df9b | ||
|
|
72547aa3bb | ||
|
|
577fd8c3c3 | ||
|
|
4f8a16df5d | ||
|
|
22356ee2b3 | ||
|
|
fa42ecd79a | ||
|
|
178c72238e | ||
|
|
e6950bdebd | ||
|
|
a774cefe85 | ||
|
|
9de17d49b7 |
@@ -0,0 +1,71 @@
|
|||||||
|
name: CI | Sync vLLM version in READMEs
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches: ["main"]
|
||||||
|
paths:
|
||||||
|
- "Dockerfile"
|
||||||
|
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: write
|
||||||
|
pull-requests: write
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
sync_version:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
name: Check README version matches Dockerfile and update if needed
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Extract vLLM version from Dockerfile and sync READMEs
|
||||||
|
run: |
|
||||||
|
echo "Extracting vLLM version from Dockerfile..."
|
||||||
|
dockerfile_version=$(grep -oP 'vllm(?:\[[\w,]+\])?==\K[\d.]+' Dockerfile | head -1)
|
||||||
|
|
||||||
|
if [ -z "$dockerfile_version" ]; then
|
||||||
|
echo "ERROR: Could not extract vLLM version from Dockerfile."
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
echo "Dockerfile vLLM version: $dockerfile_version"
|
||||||
|
echo "VLLM_VERSION=$dockerfile_version" >> $GITHUB_ENV
|
||||||
|
|
||||||
|
updated=0
|
||||||
|
for readme in README.md .runpod/README.md; do
|
||||||
|
if [ ! -f "$readme" ]; then
|
||||||
|
echo "Skipping $readme (not found)"
|
||||||
|
continue
|
||||||
|
fi
|
||||||
|
|
||||||
|
readme_version=$(grep -oP 'Current vLLM version: \[\K[\d.]+' "$readme" || echo "")
|
||||||
|
echo "$readme current version: ${readme_version:-not found}"
|
||||||
|
|
||||||
|
if [ "$readme_version" = "$dockerfile_version" ]; then
|
||||||
|
echo "$readme is already up to date."
|
||||||
|
continue
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "Updating $readme from $readme_version to $dockerfile_version..."
|
||||||
|
sed -i "s|Current vLLM version: \[${readme_version}\](https://github.com/vllm-project/vllm/releases/tag/v${readme_version})|Current vLLM version: [${dockerfile_version}](https://github.com/vllm-project/vllm/releases/tag/v${dockerfile_version})|g" "$readme"
|
||||||
|
updated=1
|
||||||
|
done
|
||||||
|
|
||||||
|
echo "UPDATED=$updated" >> $GITHUB_ENV
|
||||||
|
|
||||||
|
- name: Create Pull Request
|
||||||
|
if: env.UPDATED == '1'
|
||||||
|
uses: peter-evans/create-pull-request@v7
|
||||||
|
with:
|
||||||
|
token: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
commit-message: "docs: sync vLLM version to ${{ env.VLLM_VERSION }} in READMEs"
|
||||||
|
title: "docs: sync vLLM version to ${{ env.VLLM_VERSION }} in READMEs"
|
||||||
|
body: |
|
||||||
|
The vLLM version in the Dockerfile has been updated to `${{ env.VLLM_VERSION }}`.
|
||||||
|
|
||||||
|
This PR syncs the version badge/link in:
|
||||||
|
- `README.md`
|
||||||
|
- `.runpod/README.md`
|
||||||
|
branch: docs/sync-vllm-version-${{ env.VLLM_VERSION }}
|
||||||
|
labels: documentation
|
||||||
@@ -9,59 +9,60 @@ on:
|
|||||||
|
|
||||||
workflow_dispatch:
|
workflow_dispatch:
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: write
|
||||||
|
pull-requests: write
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
check_dep:
|
check_dep:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
name: Check python requirements file and update
|
name: Check python requirements file and update
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout
|
- name: Checkout
|
||||||
uses: actions/checkout@v2
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
- name: Check for new package version and update
|
- name: Check for new package version and update
|
||||||
run: |
|
run: |
|
||||||
echo "Fetching the current runpod version from requirements.txt..."
|
echo "Fetching current runpod version from requirements.txt..."
|
||||||
|
|
||||||
# Get current version, allowing both == and ~= in the search pattern
|
|
||||||
current_version=$(grep -oP 'runpod[~=]{1,2}\K[^"]+' ./builder/requirements.txt)
|
|
||||||
echo "Current version: $current_version"
|
|
||||||
|
|
||||||
# Extract major and minor from current version
|
# Match runpod with any version specifier or no specifier at all
|
||||||
current_major_minor=$(echo $current_version | cut -d. -f1,2)
|
current_version=$(grep -oP '^runpod([~>=!<]{1,2}\K[\d.]+)?' ./builder/requirements.txt | grep -oP '[\d.]+' || echo "")
|
||||||
echo "Current major.minor: $current_major_minor"
|
echo "Current version: ${current_version:-unset}"
|
||||||
|
|
||||||
echo "Fetching the latest runpod version from PyPI..."
|
echo "Fetching latest runpod version from PyPI..."
|
||||||
|
new_version=$(curl -sf https://pypi.org/pypi/runpod/json | jq -r .info.version)
|
||||||
# Get new version from PyPI
|
|
||||||
new_version=$(curl -s https://pypi.org/pypi/runpod/json | jq -r .info.version)
|
|
||||||
echo "NEW_VERSION_ENV=$new_version" >> $GITHUB_ENV
|
echo "NEW_VERSION_ENV=$new_version" >> $GITHUB_ENV
|
||||||
echo "New version: $new_version"
|
echo "New version: $new_version"
|
||||||
|
|
||||||
# Extract major and minor from new version
|
|
||||||
new_major_minor=$(echo $new_version | cut -d. -f1,2)
|
|
||||||
echo "New major.minor: $new_major_minor"
|
|
||||||
|
|
||||||
if [ -z "$new_version" ]; then
|
if [ -z "$new_version" ]; then
|
||||||
echo "ERROR: Failed to fetch the new version from PyPI."
|
echo "ERROR: Failed to fetch new version from PyPI."
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# Check if the major or minor version is different
|
if [ -z "$current_version" ]; then
|
||||||
if [ "$current_major_minor" = "$new_major_minor" ]; then
|
echo "No version pin found — pinning to $new_version."
|
||||||
echo "No update needed. The new version ($new_major_minor) is within the allowed range (~= $current_major_minor)."
|
else
|
||||||
|
current_major_minor=$(echo "$current_version" | cut -d. -f1,2)
|
||||||
|
new_major_minor=$(echo "$new_version" | cut -d. -f1,2)
|
||||||
|
echo "Current major.minor: $current_major_minor New major.minor: $new_major_minor"
|
||||||
|
|
||||||
|
if [ "$current_major_minor" = "$new_major_minor" ]; then
|
||||||
|
echo "No update needed. New version ($new_version) is within ~= $current_major_minor range."
|
||||||
exit 0
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "New major/minor detected ($new_major_minor). Updating requirements.txt..."
|
||||||
fi
|
fi
|
||||||
|
|
||||||
echo "New major/minor detected ($new_major_minor). Updating requirements.txt..."
|
# Replace any `runpod`, `runpod==x`, `runpod~=x`, etc. with pinned version
|
||||||
|
sed -i "s|^runpod.*|runpod~=$new_version|" ./builder/requirements.txt
|
||||||
# Update requirements.txt, preserving the existing constraint type (~= or ==)
|
echo "requirements.txt updated."
|
||||||
sed -i "s/runpod[~=][^ ]*/runpod~=$new_version/" ./builder/requirements.txt
|
|
||||||
echo "requirements.txt has been updated."
|
|
||||||
|
|
||||||
- name: Create Pull Request
|
- name: Create Pull Request
|
||||||
uses: peter-evans/create-pull-request@v3
|
uses: peter-evans/create-pull-request@v7
|
||||||
with:
|
with:
|
||||||
token: ${{ secrets.GITHUB_TOKEN }}
|
token: ${{ secrets.GITHUB_TOKEN }}
|
||||||
commit-message: Update runpod package version
|
commit-message: "chore: update runpod to ${{ env.NEW_VERSION_ENV }}"
|
||||||
title: Update runpod package version
|
title: "chore: update runpod to ${{ env.NEW_VERSION_ENV }}"
|
||||||
body: The package version has been updated to ${{ env.NEW_VERSION_ENV }}
|
body: The `runpod` package has been updated to `${{ env.NEW_VERSION_ENV }}`.
|
||||||
branch: runpod-package-update
|
branch: runpod-package-update
|
||||||
|
|||||||
@@ -3,7 +3,7 @@ name: Release
|
|||||||
on:
|
on:
|
||||||
push:
|
push:
|
||||||
tags:
|
tags:
|
||||||
- "v[0-9]+.[0-9]+.[0-9]+*" # Trigger on version tags like v1.0.0, v2.1.0, etc.
|
- "v[0-9]+.[0-9]+.[0-9]+*"
|
||||||
workflow_dispatch:
|
workflow_dispatch:
|
||||||
inputs:
|
inputs:
|
||||||
version:
|
version:
|
||||||
@@ -53,16 +53,13 @@ jobs:
|
|||||||
|
|
||||||
# Determine version based on trigger type
|
# Determine version based on trigger type
|
||||||
if [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then
|
if [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then
|
||||||
# Manual trigger: use input version
|
|
||||||
VERSION="${{ github.event.inputs.version }}"
|
VERSION="${{ github.event.inputs.version }}"
|
||||||
echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV
|
elif [[ "${{ github.event_name }}" == "release" ]]; then
|
||||||
echo "IS_MANUAL_RELEASE=true" >> $GITHUB_ENV
|
VERSION="${{ github.event.release.tag_name }}"
|
||||||
else
|
else
|
||||||
# Tag trigger: use tag name (remove refs/tags/ prefix)
|
|
||||||
VERSION=${GITHUB_REF#refs/tags/}
|
VERSION=${GITHUB_REF#refs/tags/}
|
||||||
echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV
|
|
||||||
echo "IS_MANUAL_RELEASE=false" >> $GITHUB_ENV
|
|
||||||
fi
|
fi
|
||||||
|
echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV
|
||||||
|
|
||||||
- name: Build and push the images to Docker Hub
|
- name: Build and push the images to Docker Hub
|
||||||
uses: docker/bake-action@v2
|
uses: docker/bake-action@v2
|
||||||
@@ -80,19 +77,41 @@ jobs:
|
|||||||
echo "Version: ${{ env.RELEASE_VERSION }}"
|
echo "Version: ${{ env.RELEASE_VERSION }}"
|
||||||
echo "Docker Image: ${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}"
|
echo "Docker Image: ${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}"
|
||||||
|
|
||||||
|
- name: Fetch Release Notes
|
||||||
|
run: |
|
||||||
|
RESPONSE=$(curl -sf \
|
||||||
|
-H "Authorization: token ${{ github.token }}" \
|
||||||
|
"https://api.github.com/repos/${{ github.repository }}/releases/tags/${{ env.RELEASE_VERSION }}" 2>/dev/null) || true
|
||||||
|
if [[ -n "$RESPONSE" ]]; then
|
||||||
|
NOTES=$(echo "$RESPONSE" | jq -r '.body // empty')
|
||||||
|
fi
|
||||||
|
printf '%s' "${NOTES:-No release notes available.}" > /tmp/release_notes.txt
|
||||||
|
|
||||||
- name: Notify Slack
|
- name: Notify Slack
|
||||||
run: |
|
run: |
|
||||||
curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \
|
jq -n \
|
||||||
-H "Content-Type: application/json" \
|
--arg version "${{ env.RELEASE_VERSION }}" \
|
||||||
-d '{
|
--arg docker "${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}" \
|
||||||
"text": ":rocket: New :runpod-new-whiteonpurple: Runpod worker-vllm release: *${{ env.RELEASE_VERSION }}*",
|
--rawfile notes /tmp/release_notes.txt \
|
||||||
"blocks": [
|
--arg url "https://github.com/${{ github.repository }}/releases/tag/${{ env.RELEASE_VERSION }}" \
|
||||||
|
'{
|
||||||
|
text: (":rocket: New :runpod-new-whiteonpurple: Runpod worker-vllm release: *" + $version + "*"),
|
||||||
|
blocks: [
|
||||||
{
|
{
|
||||||
"type": "section",
|
type: "section",
|
||||||
"text": {
|
text: {
|
||||||
"type": "mrkdwn",
|
type: "mrkdwn",
|
||||||
"text": ":banana-dance: *New Release — worker-vllm ${{ env.RELEASE_VERSION }}*\n*Docker:* `${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}`\n<https://github.com/${{ github.repository }}/releases/tag/${{ env.RELEASE_VERSION }}|View release on GitHub>"
|
text: (":banana-dance: *New Release — worker-vllm " + $version + "*\n*Docker:* `" + $docker + "`\n<" + $url + "|View release on GitHub>")
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
type: "section",
|
||||||
|
text: {
|
||||||
|
type: "mrkdwn",
|
||||||
|
text: ("*Release Notes:*\n" + $notes)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
}'
|
}' | curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-d @-
|
||||||
|
|||||||
@@ -6,6 +6,8 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
|
|||||||
|
|
||||||
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
||||||
|
|
||||||
|
Current vLLM version: [0.22.1](https://github.com/vllm-project/vllm/releases/tag/v0.22.1)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## Endpoint Configuration
|
## Endpoint Configuration
|
||||||
@@ -27,9 +29,21 @@ All behaviour is controlled through environment variables:
|
|||||||
| `REASONING_PARSER` | Parser for reasoning-capable models | | "deepseek_r1", "qwen3", "granite", "hunyuan_a13b" |
|
| `REASONING_PARSER` | Parser for reasoning-capable models | | "deepseek_r1", "qwen3", "granite", "hunyuan_a13b" |
|
||||||
| `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | Override served model name in API | | String |
|
| `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | Override served model name in API | | String |
|
||||||
| `MAX_CONCURRENCY` | Maximum concurrent requests | 300 | Integer |
|
| `MAX_CONCURRENCY` | Maximum concurrent requests | 300 | Integer |
|
||||||
|
| `ENFORCE_EAGER` | If True, we will disable CUDA graph and always execute the model in eager mode. If False, we will use CUDA graph and eager execution in hybrid for maximal performance and flexibility. | true | boolean (true or false) |
|
||||||
|
|
||||||
**Pass any vLLM engine arg** not listed above by setting an env var with the **UPPERCASED** field name (e.g. `MAX_MODEL_LEN=4096`, `ENABLE_CHUNKED_PREFILL=true`). The worker auto-discovers all `AsyncEngineArgs` fields from env. See the [vLLM engine args docs](https://docs.vllm.ai/en/latest/configuration/engine_args) for all available options.
|
**Pass any vLLM engine arg** not listed above by setting an env var with the **UPPERCASED** field name (e.g. `MAX_MODEL_LEN=4096`, `ENABLE_CHUNKED_PREFILL=true`). The worker auto-discovers all `AsyncEngineArgs` fields from env. See the [vLLM engine args docs](https://docs.vllm.ai/en/latest/configuration/engine_args) for all available options.
|
||||||
|
|
||||||
|
**Configuration file:** You can also supply a `config.yaml` instead of (or alongside) env vars. Mount it at `/vllm_config.yaml` in the container, or set `VLLM_CONFIG_FILE` to a custom path. Use the same key names as `vllm serve` — hyphens and underscores both work:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model: meta-llama/Llama-3.1-8B-Instruct
|
||||||
|
max-model-len: 8192
|
||||||
|
gpu-memory-utilization: 0.90
|
||||||
|
quantization: awq
|
||||||
|
```
|
||||||
|
|
||||||
|
Environment variables always override config file values.
|
||||||
|
|
||||||
For complete configuration options, see the [full configuration documentation](https://github.com/runpod-workers/worker-vllm/blob/main/docs/configuration.md).
|
For complete configuration options, see the [full configuration documentation](https://github.com/runpod-workers/worker-vllm/blob/main/docs/configuration.md).
|
||||||
|
|
||||||
### Specify Transformers Version
|
### Specify Transformers Version
|
||||||
|
|||||||
+22
-2
@@ -9,7 +9,7 @@
|
|||||||
"containerDiskInGb": 150,
|
"containerDiskInGb": 150,
|
||||||
"gpuIds": "ADA_80_PRO,AMPERE_80",
|
"gpuIds": "ADA_80_PRO,AMPERE_80",
|
||||||
"gpuCount": 1,
|
"gpuCount": 1,
|
||||||
"allowedCudaVersions": ["12.9", "12.8"],
|
"allowedCudaVersions": ["13.0"],
|
||||||
"presets": [
|
"presets": [
|
||||||
{
|
{
|
||||||
"name": "deepseek-ai/deepseek-r1-distill-llama-8b",
|
"name": "deepseek-ai/deepseek-r1-distill-llama-8b",
|
||||||
@@ -621,7 +621,7 @@
|
|||||||
"name": "Enforce Eager",
|
"name": "Enforce Eager",
|
||||||
"type": "boolean",
|
"type": "boolean",
|
||||||
"description": "Always use eager-mode PyTorch. If False (0), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility",
|
"description": "Always use eager-mode PyTorch. If False (0), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility",
|
||||||
"default": false,
|
"default": true,
|
||||||
"advanced": true
|
"advanced": true
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
@@ -795,6 +795,26 @@
|
|||||||
"default": "",
|
"default": "",
|
||||||
"advanced": true
|
"advanced": true
|
||||||
}
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"key": "PYTORCH_ALLOC_CONF",
|
||||||
|
"input": {
|
||||||
|
"name": "PyTorch Alloc Config",
|
||||||
|
"type": "string",
|
||||||
|
"description": "PyTorch allocation configuration, remove this if you want to use the default configuration",
|
||||||
|
"default": "expandable_segments:True",
|
||||||
|
"advanced": true
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"key": "VLLM_USE_DEEP_GEMM",
|
||||||
|
"input": {
|
||||||
|
"name": "Use DeepGEMM",
|
||||||
|
"type": "string",
|
||||||
|
"description": "Enable DeepGEMM FP8 kernels (MoE and MQA logits). Set to 1 to enable, 0 to disable. Required for DeepSeek V4 models. Disabled by default — enable on H100/H200 for potential throughput gains. Some GPUs (e.g. H20) may perform better with this off.",
|
||||||
|
"default": "0",
|
||||||
|
"advanced": true
|
||||||
|
}
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -5,7 +5,7 @@
|
|||||||
"input": {
|
"input": {
|
||||||
"prompt": "Write a short poem about artificial intelligence."
|
"prompt": "Write a short poem about artificial intelligence."
|
||||||
},
|
},
|
||||||
"timeout": 30000
|
"timeout": 300000
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"name": "openai_messages_test",
|
"name": "openai_messages_test",
|
||||||
@@ -26,11 +26,11 @@
|
|||||||
"temperature": 0.1
|
"temperature": 0.1
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
"timeout": 30000
|
"timeout": 300000
|
||||||
}
|
}
|
||||||
],
|
],
|
||||||
"config": {
|
"config": {
|
||||||
"gpuTypeId": "NVIDIA GeForce RTX 4090",
|
"gpuTypeId": "NNVIDIA L40",
|
||||||
"gpuCount": 1,
|
"gpuCount": 1,
|
||||||
"env": [
|
"env": [
|
||||||
{
|
{
|
||||||
@@ -38,6 +38,6 @@
|
|||||||
"value": "HuggingFaceTB/SmolLM2-135M-Instruct"
|
"value": "HuggingFaceTB/SmolLM2-135M-Instruct"
|
||||||
}
|
}
|
||||||
],
|
],
|
||||||
"allowedCudaVersions": ["12.9", "12.8", "12.7", "12.6", "12.5"]
|
"allowedCudaVersions": ["13.0"]
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
+20
-6
@@ -1,16 +1,28 @@
|
|||||||
FROM nvidia/cuda:12.9.1-base-ubuntu22.04
|
FROM nvidia/cuda:13.0.2-devel-ubuntu22.04
|
||||||
|
|
||||||
RUN apt-get update -y \
|
RUN apt-get update -y \
|
||||||
&& apt-get install -y python3-pip curl \
|
&& apt-get install -y python3-pip curl git \
|
||||||
&& curl -LsSf https://astral.sh/uv/install.sh | sh
|
&& curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||||
|
|
||||||
ENV PATH="/root/.local/bin:$PATH"
|
ENV PATH="/root/.local/bin:$PATH"
|
||||||
|
|
||||||
RUN ldconfig /usr/local/cuda-12.9/compat/
|
RUN ldconfig /usr/local/cuda-13.0/compat/
|
||||||
|
|
||||||
# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels
|
# nixl_ep PyPI wheels are compiled against CUDA 12.x and require libcudart.so.12.
|
||||||
|
# CUDA 13 runtime is ABI-compatible with CUDA 12, so symlinking is safe.
|
||||||
|
# Symlink into /usr/local/cuda/lib64 (already in LD_LIBRARY_PATH) so the linker
|
||||||
|
# finds it by filename scan rather than relying on ldcache SONAME lookup.
|
||||||
|
RUN ln -sf /usr/local/cuda/lib64/libcudart.so.13 /usr/local/cuda/lib64/libcudart.so.12 && ldconfig
|
||||||
|
|
||||||
|
# CUDA 13.0 containers return libs to /usr/local/nvidia/lib64 so container
|
||||||
|
# providers (RunPod, Lambda, etc.) can mount host drivers there consistently.
|
||||||
|
# See: https://github.com/vllm-project/vllm/issues/18859
|
||||||
|
ENV LD_LIBRARY_PATH=/usr/local/nvidia/lib64:/usr/local/cuda/lib64:$LD_LIBRARY_PATH
|
||||||
|
|
||||||
|
# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
|
||||||
RUN uv pip install --system "packaging>=24.2" && \
|
RUN uv pip install --system "packaging>=24.2" && \
|
||||||
uv pip install --system "vllm[flashinfer]==0.17.1" --extra-index-url https://download.pytorch.org/whl/cu129
|
uv pip install --system "vllm[flashinfer]==0.22.1" && \
|
||||||
|
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation
|
||||||
|
|
||||||
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
||||||
COPY builder/requirements.txt /requirements.txt
|
COPY builder/requirements.txt /requirements.txt
|
||||||
@@ -42,7 +54,9 @@ ENV MODEL_NAME=$MODEL_NAME \
|
|||||||
# Prevent rayon thread pool panic in containers where ulimit -u < nproc
|
# Prevent rayon thread pool panic in containers where ulimit -u < nproc
|
||||||
# (tokenizers uses Rust's rayon which tries to spawn threads = CPU cores)
|
# (tokenizers uses Rust's rayon which tries to spawn threads = CPU cores)
|
||||||
TOKENIZERS_PARALLELISM=false \
|
TOKENIZERS_PARALLELISM=false \
|
||||||
RAYON_NUM_THREADS=4
|
RAYON_NUM_THREADS=4 \
|
||||||
|
# Disable DeepGEMM MoE kernels by default; override with VLLM_USE_DEEP_GEMM=1 to enable
|
||||||
|
VLLM_USE_DEEP_GEMM=0
|
||||||
|
|
||||||
ENV PYTHONPATH="/:/vllm-workspace"
|
ENV PYTHONPATH="/:/vllm-workspace"
|
||||||
|
|
||||||
|
|||||||
@@ -8,7 +8,8 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
|
|||||||
|
|
||||||

|

|
||||||
|
|
||||||
Current vLLM version: [0.16.0](https://github.com/vllm-project/vllm/releases/tag/v0.16.0)
|
Current vLLM version: [0.22.1](https://github.com/vllm-project/vllm/releases/tag/v0.22.1)
|
||||||
|
|
||||||
|
|
||||||
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
|
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
|
||||||
|
|
||||||
@@ -46,7 +47,7 @@ Current vLLM version: [0.16.0](https://github.com/vllm-project/vllm/releases/tag
|
|||||||
**📦 Docker Image**: `runpod/worker-v1-vllm:<version>`
|
**📦 Docker Image**: `runpod/worker-v1-vllm:<version>`
|
||||||
|
|
||||||
- **Available Versions**: See [GitHub Releases](https://github.com/runpod-workers/worker-vllm/releases)
|
- **Available Versions**: See [GitHub Releases](https://github.com/runpod-workers/worker-vllm/releases)
|
||||||
- **CUDA Compatibility**: Requires CUDA >= 12.1
|
- **CUDA Compatibility**: Requires CUDA >= 13.0
|
||||||
|
|
||||||
### Configuration
|
### Configuration
|
||||||
|
|
||||||
@@ -77,6 +78,20 @@ Configure worker-vllm using environment variables:
|
|||||||
|
|
||||||
Any env var whose name matches a valid `AsyncEngineArgs` field (uppercased) is applied automatically. Backward-compat aliases: `MODEL_NAME`, `TOKENIZER_NAME`, `MAX_CONTEXT_LEN_TO_CAPTURE`. This lets you configure any vLLM option without waiting for explicit worker support.
|
Any env var whose name matches a valid `AsyncEngineArgs` field (uppercased) is applied automatically. Backward-compat aliases: `MODEL_NAME`, `TOKENIZER_NAME`, `MAX_CONTEXT_LEN_TO_CAPTURE`. This lets you configure any vLLM option without waiting for explicit worker support.
|
||||||
|
|
||||||
|
### Configuration File (config.yaml)
|
||||||
|
|
||||||
|
As an alternative to environment variables, you can supply a `config.yaml` file using the same key names as `vllm serve` (hyphens or underscores both work):
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model: meta-llama/Llama-3.1-8B-Instruct
|
||||||
|
max-model-len: 8192
|
||||||
|
gpu-memory-utilization: 0.90
|
||||||
|
quantization: awq
|
||||||
|
tensor-parallel-size: 2
|
||||||
|
```
|
||||||
|
|
||||||
|
Mount the file into the container at `/vllm_config.yaml`, or point to a custom path with the `VLLM_CONFIG_FILE` env var. Environment variables always take precedence over config file values.
|
||||||
|
|
||||||
For the complete list of all available environment variables, examples, and detailed descriptions: **[Configuration](docs/configuration.md)**
|
For the complete list of all available environment variables, examples, and detailed descriptions: **[Configuration](docs/configuration.md)**
|
||||||
|
|
||||||
### Specify Transformers Version
|
### Specify Transformers Version
|
||||||
|
|||||||
@@ -1,15 +1,15 @@
|
|||||||
ray
|
ray
|
||||||
pandas
|
pandas
|
||||||
pyarrow
|
pyarrow
|
||||||
runpod
|
runpod==1.9.1
|
||||||
huggingface-hub
|
huggingface-hub
|
||||||
lmcache==0.4.2
|
lmcache==0.4.6
|
||||||
packaging>=24.2
|
packaging>=24.2
|
||||||
typing-extensions>=4.8.0
|
typing-extensions>=4.8.0
|
||||||
pydantic
|
pydantic
|
||||||
pydantic-settings
|
pydantic-settings
|
||||||
hf-transfer
|
hf-transfer
|
||||||
transformers>=4.57.0,<5
|
transformers>=5
|
||||||
bitsandbytes>=0.45.0
|
bitsandbytes>=0.45.0
|
||||||
kernels
|
kernels<0.15
|
||||||
torch-c-dlpack-ext
|
torch-c-dlpack-ext
|
||||||
|
|||||||
@@ -0,0 +1,10 @@
|
|||||||
|
model: meta-llama/Llama-3.1-8B-Instruct
|
||||||
|
gpu-memory-utilization: 0.95
|
||||||
|
max-model-len: 8192
|
||||||
|
dtype: auto
|
||||||
|
trust-remote-code: true
|
||||||
|
quantization: fp8
|
||||||
|
kv-cache-dtype: fp8
|
||||||
|
enforce-eager: false
|
||||||
|
enable-prefix-caching: true
|
||||||
|
speculative-config: '{"model":"RedHatAI/Llama-3.1-8B-Instruct-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||||
@@ -0,0 +1,10 @@
|
|||||||
|
model: Qwen/Qwen3-8B
|
||||||
|
gpu-memory-utilization: 0.95
|
||||||
|
max-model-len: 8192
|
||||||
|
dtype: auto
|
||||||
|
trust-remote-code: true
|
||||||
|
quantization: fp8
|
||||||
|
kv-cache-dtype: fp8
|
||||||
|
enforce-eager: false
|
||||||
|
enable-prefix-caching: true
|
||||||
|
speculative-config: '{"model":"RedHatAI/Qwen3-8B-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||||
@@ -97,10 +97,13 @@ If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When
|
|||||||
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
|
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
|
||||||
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
|
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
|
||||||
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. |
|
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models. |
|
||||||
|
| `VLLM_USE_DEEP_GEMM` | `0` | `str` (`0`/`1`) | Enable DeepGEMM FP8 kernels for MoE and MQA logits computation. Disabled by default. Must be `"0"` or `"1"` — not `true`/`false`. See note below. |
|
||||||
| `ATTENTION_BACKEND` | `None` | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `FLASHINFER`, `TRITON_FLASH_ATTN`). Replaces deprecated `VLLM_ATTENTION_BACKEND`. |
|
| `ATTENTION_BACKEND` | `None` | `str` | Attention backend to use (e.g., `FLASH_ATTN`, `FLASHINFER`, `TRITON_FLASH_ATTN`). Replaces deprecated `VLLM_ATTENTION_BACKEND`. |
|
||||||
| `ASYNC_SCHEDULING` | `None` | `bool` | Enable async scheduling (overlaps engine scheduling with GPU execution). Default: enabled in vLLM 0.14.0+. Set to `false` to disable. |
|
| `ASYNC_SCHEDULING` | `None` | `bool` | Enable async scheduling (overlaps engine scheduling with GPU execution). Default: enabled in vLLM 0.14.0+. Set to `false` to disable. |
|
||||||
| `STREAM_INTERVAL` | `1` | `int` | Controls how often to yield streaming results. Lower = more frequent updates. |
|
| `STREAM_INTERVAL` | `1` | `int` | Controls how often to yield streaming results. Lower = more frequent updates. |
|
||||||
|
|
||||||
|
> **Note (`VLLM_USE_DEEP_GEMM`):** DeepGEMM is used in two places: MoE weight computation and MQA logits computation. It is necessary for MQA logits computation on supported hardware — required for DeepSeek V4 models. Set `VLLM_USE_DEEP_GEMM=1` to enable. Set `VLLM_USE_DEEP_GEMM=0` to disable the MoE part and fall back to flashinfer/cutlass FP8 kernels. **Value must be `"0"` or `"1"` — not `"true"`/`"false"`.** Some users report better performance with `VLLM_USE_DEEP_GEMM=0`, particularly on H20 GPUs. Disabling it also skips the DeepGEMM warmup phase, reducing cold-start time. Requires CUDA 13.0+ and SM90+ (H100/H200) to use; the library is installed but inactive by default.
|
||||||
|
|
||||||
## Tokenizer Settings
|
## Tokenizer Settings
|
||||||
|
|
||||||
| Variable | Default | Type/Choices | Description |
|
| Variable | Default | Type/Choices | Description |
|
||||||
|
|||||||
+96
-32
@@ -1,4 +1,5 @@
|
|||||||
import asyncio
|
import asyncio
|
||||||
|
import inspect
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
@@ -7,6 +8,7 @@ from typing import AsyncGenerator, Optional
|
|||||||
|
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
from vllm import AsyncLLMEngine
|
from vllm import AsyncLLMEngine
|
||||||
|
from vllm.inputs import TextPrompt
|
||||||
from vllm.entrypoints.logger import RequestLogger
|
from vllm.entrypoints.logger import RequestLogger
|
||||||
from vllm.entrypoints.anthropic.protocol import AnthropicMessagesRequest, AnthropicMessagesResponse, AnthropicError, AnthropicErrorResponse
|
from vllm.entrypoints.anthropic.protocol import AnthropicMessagesRequest, AnthropicMessagesResponse, AnthropicError, AnthropicErrorResponse
|
||||||
from vllm.entrypoints.anthropic.serving import AnthropicServingMessages
|
from vllm.entrypoints.anthropic.serving import AnthropicServingMessages
|
||||||
@@ -19,6 +21,7 @@ from vllm.entrypoints.openai.models.protocol import BaseModelPath, LoRAModulePat
|
|||||||
from vllm.entrypoints.openai.models.serving import OpenAIServingModels
|
from vllm.entrypoints.openai.models.serving import OpenAIServingModels
|
||||||
from vllm.entrypoints.openai.responses.protocol import ResponsesRequest, ResponsesResponse
|
from vllm.entrypoints.openai.responses.protocol import ResponsesRequest, ResponsesResponse
|
||||||
from vllm.entrypoints.openai.responses.serving import OpenAIServingResponses
|
from vllm.entrypoints.openai.responses.serving import OpenAIServingResponses
|
||||||
|
from vllm.entrypoints.serve.render.serving import OpenAIServingRender
|
||||||
|
|
||||||
from constants import DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MAX_CONCURRENCY, DEFAULT_MIN_BATCH_SIZE
|
from constants import DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MAX_CONCURRENCY, DEFAULT_MIN_BATCH_SIZE
|
||||||
from engine_args import get_engine_args
|
from engine_args import get_engine_args
|
||||||
@@ -29,20 +32,33 @@ class vLLMEngine:
|
|||||||
def __init__(self, engine = None):
|
def __init__(self, engine = None):
|
||||||
load_dotenv() # For local development
|
load_dotenv() # For local development
|
||||||
self.engine_args = get_engine_args()
|
self.engine_args = get_engine_args()
|
||||||
logging.info(f"Engine args: {self.engine_args}")
|
|
||||||
|
if engine is None:
|
||||||
# Initialize vLLM engine first
|
ea = self.engine_args
|
||||||
self.llm = self._initialize_llm() if engine is None else engine.llm
|
summary = {
|
||||||
|
"model": ea.model,
|
||||||
# Only create custom tokenizer wrapper if not using mistral tokenizer mode
|
"dtype": ea.dtype,
|
||||||
# For mistral models, let vLLM handle tokenizer initialization
|
"quantization": ea.quantization,
|
||||||
if self.engine_args.tokenizer_mode != 'mistral':
|
"max_model_len": ea.max_model_len,
|
||||||
self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model,
|
"tensor_parallel_size": ea.tensor_parallel_size,
|
||||||
self.engine_args.tokenizer_revision,
|
"gpu_memory_utilization": ea.gpu_memory_utilization,
|
||||||
self.engine_args.trust_remote_code)
|
}
|
||||||
|
if ea.tokenizer and ea.tokenizer != ea.model:
|
||||||
|
summary["tokenizer"] = ea.tokenizer
|
||||||
|
logging.info("Engine config: %s", summary)
|
||||||
|
logging.debug("Full engine args: %s", ea)
|
||||||
|
|
||||||
|
self.llm = self._initialize_llm()
|
||||||
|
|
||||||
|
if self.engine_args.tokenizer_mode != 'mistral':
|
||||||
|
self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model,
|
||||||
|
self.engine_args.tokenizer_revision,
|
||||||
|
self.engine_args.trust_remote_code)
|
||||||
|
else:
|
||||||
|
self.tokenizer = None
|
||||||
else:
|
else:
|
||||||
# For mistral models, we'll get the tokenizer from vLLM later
|
self.llm = engine.llm
|
||||||
self.tokenizer = None
|
self.tokenizer = engine.tokenizer
|
||||||
|
|
||||||
self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
|
self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
|
||||||
self.default_batch_size = int(os.getenv("DEFAULT_BATCH_SIZE", DEFAULT_BATCH_SIZE))
|
self.default_batch_size = int(os.getenv("DEFAULT_BATCH_SIZE", DEFAULT_BATCH_SIZE))
|
||||||
@@ -115,7 +131,7 @@ class vLLMEngine:
|
|||||||
if apply_chat_template or isinstance(llm_input, list):
|
if apply_chat_template or isinstance(llm_input, list):
|
||||||
tokenizer_wrapper = self._get_tokenizer_for_chat_template()
|
tokenizer_wrapper = self._get_tokenizer_for_chat_template()
|
||||||
llm_input = tokenizer_wrapper.apply_chat_template(llm_input)
|
llm_input = tokenizer_wrapper.apply_chat_template(llm_input)
|
||||||
results_generator = self.llm.generate(llm_input, validated_sampling_params, request_id)
|
results_generator = self.llm.generate(TextPrompt(prompt=llm_input), validated_sampling_params, request_id)
|
||||||
n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True
|
n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True
|
||||||
last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0}
|
last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0}
|
||||||
|
|
||||||
@@ -205,19 +221,48 @@ class OpenAIvLLMEngine(vLLMEngine):
|
|||||||
self.raw_openai_output = bool(int(raw_output_env))
|
self.raw_openai_output = bool(int(raw_output_env))
|
||||||
|
|
||||||
def _load_lora_adapters(self):
|
def _load_lora_adapters(self):
|
||||||
adapters = []
|
lora_modules_env = os.getenv("LORA_MODULES", "")
|
||||||
try:
|
if not lora_modules_env:
|
||||||
adapters = json.loads(os.getenv("LORA_MODULES", '[]'))
|
return []
|
||||||
except Exception as e:
|
|
||||||
logging.info(f"---Initialized adapter json load error: {e}")
|
|
||||||
|
|
||||||
for i, adapter in enumerate(adapters):
|
try:
|
||||||
|
parsed = json.loads(lora_modules_env)
|
||||||
|
except json.JSONDecodeError as e:
|
||||||
|
logging.error(
|
||||||
|
"LORA_MODULES could not be parsed as JSON: %s — no LoRA adapters loaded. Value: %r",
|
||||||
|
e, lora_modules_env,
|
||||||
|
)
|
||||||
|
return []
|
||||||
|
|
||||||
|
# Accept a single adapter dict as well as an array
|
||||||
|
if isinstance(parsed, dict):
|
||||||
|
parsed = [parsed]
|
||||||
|
|
||||||
|
if not isinstance(parsed, list):
|
||||||
|
logging.error(
|
||||||
|
"LORA_MODULES must be a JSON array of adapter objects, got %s — no LoRA adapters loaded.",
|
||||||
|
type(parsed).__name__,
|
||||||
|
)
|
||||||
|
return []
|
||||||
|
|
||||||
|
adapters = []
|
||||||
|
for i, adapter in enumerate(parsed):
|
||||||
try:
|
try:
|
||||||
adapters[i] = LoRAModulePath(**adapter)
|
adapters.append(LoRAModulePath(**adapter))
|
||||||
logging.info(f"---Initialized adapter: {adapter}")
|
logging.info("Loaded LoRA adapter config [%d]: %s", i, adapter)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logging.info(f"---Initialized adapter not worked: {e}")
|
logging.error(
|
||||||
continue
|
"Failed to parse LoRA adapter at index %d: %s. Config: %r",
|
||||||
|
i, e, adapter,
|
||||||
|
)
|
||||||
|
|
||||||
|
if parsed and not adapters:
|
||||||
|
logging.error(
|
||||||
|
"LORA_MODULES specified %d adapter(s) but none could be loaded — "
|
||||||
|
"OpenAI model name lookups for LoRA adapters will fail.",
|
||||||
|
len(parsed),
|
||||||
|
)
|
||||||
|
|
||||||
return adapters
|
return adapters
|
||||||
|
|
||||||
async def _ensure_engines_initialized(self):
|
async def _ensure_engines_initialized(self):
|
||||||
@@ -246,16 +291,32 @@ class OpenAIvLLMEngine(vLLMEngine):
|
|||||||
lora_modules=self.lora_adapters,
|
lora_modules=self.lora_adapters,
|
||||||
)
|
)
|
||||||
await self.serving_models.init_static_loras()
|
await self.serving_models.init_static_loras()
|
||||||
|
|
||||||
# Get chat template from vLLM tokenizer if available
|
# Get chat template from vLLM tokenizer if available
|
||||||
chat_template = None
|
chat_template = None
|
||||||
if self.tokenizer and hasattr(self.tokenizer, 'tokenizer'):
|
if self.tokenizer and hasattr(self.tokenizer, 'tokenizer'):
|
||||||
chat_template = self.tokenizer.tokenizer.chat_template
|
chat_template = self.tokenizer.tokenizer.chat_template
|
||||||
|
|
||||||
|
self.openai_serving_render = OpenAIServingRender(
|
||||||
|
model_config=self.llm.model_config,
|
||||||
|
renderer=self.llm.renderer,
|
||||||
|
model_registry=self.serving_models.registry,
|
||||||
|
request_logger=None,
|
||||||
|
chat_template=chat_template,
|
||||||
|
chat_template_content_format="auto",
|
||||||
|
trust_request_chat_template=os.getenv('TRUST_REQUEST_CHAT_TEMPLATE', 'false').lower() == 'true',
|
||||||
|
enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true',
|
||||||
|
exclude_tools_when_tool_choice_none=os.getenv('EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE', 'false').lower() == 'true',
|
||||||
|
tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None,
|
||||||
|
reasoning_parser=os.getenv('REASONING_PARSER', "") or None,
|
||||||
|
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
|
||||||
|
)
|
||||||
|
|
||||||
self.chat_engine = OpenAIServingChat(
|
self.chat_engine = OpenAIServingChat(
|
||||||
engine_client=self.llm,
|
engine_client=self.llm,
|
||||||
models=self.serving_models,
|
models=self.serving_models,
|
||||||
response_role=self.response_role,
|
response_role=self.response_role,
|
||||||
|
openai_serving_render=self.openai_serving_render,
|
||||||
request_logger=None,
|
request_logger=None,
|
||||||
chat_template=chat_template,
|
chat_template=chat_template,
|
||||||
chat_template_content_format="auto",
|
chat_template_content_format="auto",
|
||||||
@@ -268,20 +329,20 @@ class OpenAIvLLMEngine(vLLMEngine):
|
|||||||
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
|
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
|
||||||
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
||||||
enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true',
|
enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true',
|
||||||
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
|
|
||||||
)
|
)
|
||||||
self.completion_engine = OpenAIServingCompletion(
|
self.completion_engine = OpenAIServingCompletion(
|
||||||
engine_client=self.llm,
|
engine_client=self.llm,
|
||||||
models=self.serving_models,
|
models=self.serving_models,
|
||||||
|
openai_serving_render=self.openai_serving_render,
|
||||||
request_logger=None,
|
request_logger=None,
|
||||||
return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true',
|
return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true',
|
||||||
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
|
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
|
||||||
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
||||||
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
|
|
||||||
)
|
)
|
||||||
self.responses_engine = OpenAIServingResponses(
|
self.responses_engine = OpenAIServingResponses(
|
||||||
engine_client=self.llm,
|
engine_client=self.llm,
|
||||||
models=self.serving_models,
|
models=self.serving_models,
|
||||||
|
openai_serving_render=self.openai_serving_render,
|
||||||
request_logger=None,
|
request_logger=None,
|
||||||
chat_template=chat_template,
|
chat_template=chat_template,
|
||||||
chat_template_content_format="auto",
|
chat_template_content_format="auto",
|
||||||
@@ -293,12 +354,12 @@ class OpenAIvLLMEngine(vLLMEngine):
|
|||||||
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
|
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
|
||||||
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
||||||
enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true',
|
enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true',
|
||||||
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
|
|
||||||
)
|
)
|
||||||
self.messages_engine = AnthropicServingMessages(
|
self.messages_engine = AnthropicServingMessages(
|
||||||
engine_client=self.llm,
|
engine_client=self.llm,
|
||||||
models=self.serving_models,
|
models=self.serving_models,
|
||||||
response_role=self.response_role,
|
response_role=self.response_role,
|
||||||
|
openai_serving_render=self.openai_serving_render,
|
||||||
request_logger=None,
|
request_logger=None,
|
||||||
chat_template=chat_template,
|
chat_template=chat_template,
|
||||||
chat_template_content_format="auto",
|
chat_template_content_format="auto",
|
||||||
@@ -310,8 +371,11 @@ class OpenAIvLLMEngine(vLLMEngine):
|
|||||||
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
||||||
)
|
)
|
||||||
|
|
||||||
if hasattr(self.chat_engine, 'warmup'):
|
warmup = getattr(self.chat_engine, 'warmup', None)
|
||||||
await self.chat_engine.warmup()
|
if callable(warmup):
|
||||||
|
result = warmup()
|
||||||
|
if inspect.isawaitable(result):
|
||||||
|
await result
|
||||||
|
|
||||||
async def generate(self, openai_request: JobInput):
|
async def generate(self, openai_request: JobInput):
|
||||||
# Ensure engines are ready (no-op if already initialized at startup)
|
# Ensure engines are ready (no-op if already initialized at startup)
|
||||||
|
|||||||
@@ -82,6 +82,16 @@ def _convert_env_value_to_field_type(value: str, field_name: str, field_type: ty
|
|||||||
if type(None) in (args or ()):
|
if type(None) in (args or ()):
|
||||||
return None
|
return None
|
||||||
raise ValueError("empty value not allowed for non-optional field")
|
raise ValueError("empty value not allowed for non-optional field")
|
||||||
|
|
||||||
|
# Union[bool, str, ...]: only coerce to bool for unambiguous literals;
|
||||||
|
# otherwise preserve the string (e.g. hf_token="hf_abc..." must stay a str).
|
||||||
|
if get_origin(field_type) is not None:
|
||||||
|
union_types = [a for a in (get_args(field_type) or ()) if a is not type(None)]
|
||||||
|
if bool in union_types and str in union_types:
|
||||||
|
if str(val).lower() in ("true", "false", "1", "0", "yes", "no", "on", "off"):
|
||||||
|
return str(val).lower() in ("true", "1", "yes", "on")
|
||||||
|
return str(val)
|
||||||
|
|
||||||
effective_type = _resolve_field_type(field_type)
|
effective_type = _resolve_field_type(field_type)
|
||||||
# bool
|
# bool
|
||||||
if effective_type is bool:
|
if effective_type is bool:
|
||||||
@@ -342,6 +352,75 @@ def _sanitize_hf_overrides(hf_overrides: dict) -> dict | None:
|
|||||||
return result or None
|
return result or None
|
||||||
|
|
||||||
|
|
||||||
|
def _resolve_cached_model_path(model_name: str) -> str:
|
||||||
|
"""Return a local snapshot path when the HF cache was stored with lowercase names.
|
||||||
|
|
||||||
|
Some model stores (e.g. RunPod pre-cached volumes) normalize repo IDs to
|
||||||
|
lowercase. HuggingFace Hub stores caches as
|
||||||
|
``models--{org}--{model}/snapshots/{hash}/`` preserving the original casing,
|
||||||
|
so MODEL_NAME=Qwen/Qwen2.5-Coder-32B-Instruct-AWQ will miss a cache stored
|
||||||
|
as ``models--qwen--qwen2.5-coder-32b-instruct-awq/``.
|
||||||
|
|
||||||
|
If the exact-case cache directory is absent but a lowercase variant exists,
|
||||||
|
the latest snapshot path is returned so vLLM loads from disk rather than
|
||||||
|
attempting a redundant download.
|
||||||
|
"""
|
||||||
|
if os.path.isabs(model_name):
|
||||||
|
return model_name
|
||||||
|
|
||||||
|
cache_dir = (
|
||||||
|
os.getenv("HUGGINGFACE_HUB_CACHE")
|
||||||
|
or os.getenv("HF_HOME")
|
||||||
|
or os.path.expanduser("~/.cache/huggingface/hub")
|
||||||
|
)
|
||||||
|
|
||||||
|
folder_name = f"models--{model_name.replace('/', '--')}"
|
||||||
|
|
||||||
|
if os.path.isdir(os.path.join(cache_dir, folder_name)):
|
||||||
|
return model_name
|
||||||
|
|
||||||
|
lower_dir = os.path.join(cache_dir, folder_name.lower())
|
||||||
|
if not os.path.isdir(lower_dir):
|
||||||
|
return model_name
|
||||||
|
|
||||||
|
snapshots_dir = os.path.join(lower_dir, "snapshots")
|
||||||
|
if not os.path.isdir(snapshots_dir):
|
||||||
|
return model_name
|
||||||
|
|
||||||
|
try:
|
||||||
|
snapshots = sorted(os.listdir(snapshots_dir))
|
||||||
|
except OSError:
|
||||||
|
return model_name
|
||||||
|
|
||||||
|
if not snapshots:
|
||||||
|
return model_name
|
||||||
|
|
||||||
|
resolved = os.path.join(snapshots_dir, snapshots[-1])
|
||||||
|
logging.info(
|
||||||
|
"MODEL_NAME %r not found at original casing in HF cache; "
|
||||||
|
"resolved to lowercase cached snapshot at %r",
|
||||||
|
model_name, resolved,
|
||||||
|
)
|
||||||
|
return resolved
|
||||||
|
|
||||||
|
|
||||||
|
def _get_args_from_config_file() -> dict:
|
||||||
|
"""Load engine args from a vLLM-style config.yaml.
|
||||||
|
|
||||||
|
Checks VLLM_CONFIG_FILE env var, then falls back to /vllm_config.yaml.
|
||||||
|
Keys use the same long-form names as vllm serve (hyphens converted to underscores).
|
||||||
|
"""
|
||||||
|
import yaml
|
||||||
|
path = os.getenv("VLLM_CONFIG_FILE", "/vllm_config.yaml")
|
||||||
|
if not os.path.exists(path):
|
||||||
|
return {}
|
||||||
|
with open(path) as f:
|
||||||
|
raw = yaml.safe_load(f) or {}
|
||||||
|
normalized = {k.replace("-", "_"): v for k, v in raw.items()}
|
||||||
|
logging.info("Loaded engine args from config file %s: %s", path, list(normalized.keys()))
|
||||||
|
return normalized
|
||||||
|
|
||||||
|
|
||||||
def get_local_args():
|
def get_local_args():
|
||||||
"""
|
"""
|
||||||
Retrieve local arguments from a JSON file.
|
Retrieve local arguments from a JSON file.
|
||||||
@@ -367,6 +446,9 @@ def get_engine_args():
|
|||||||
# Start with worker custom defaults (only where we differ from vLLM)
|
# Start with worker custom defaults (only where we differ from vLLM)
|
||||||
args = dict(DEFAULT_ARGS)
|
args = dict(DEFAULT_ARGS)
|
||||||
|
|
||||||
|
# Config file values sit above defaults but below env vars
|
||||||
|
args.update(_get_args_from_config_file())
|
||||||
|
|
||||||
# Auto-discover: every AsyncEngineArgs field from env UPPERCASED (e.g. MAX_MODEL_LEN)
|
# Auto-discover: every AsyncEngineArgs field from env UPPERCASED (e.g. MAX_MODEL_LEN)
|
||||||
args.update(_get_args_from_env_auto_discover())
|
args.update(_get_args_from_env_auto_discover())
|
||||||
|
|
||||||
@@ -517,4 +599,8 @@ def get_engine_args():
|
|||||||
if speculative_config:
|
if speculative_config:
|
||||||
args["speculative_config"] = speculative_config
|
args["speculative_config"] = speculative_config
|
||||||
|
|
||||||
|
# Resolve lowercase HF cache paths (FDE-174)
|
||||||
|
if args.get("model"):
|
||||||
|
args["model"] = _resolve_cached_model_path(args["model"])
|
||||||
|
|
||||||
return AsyncEngineArgs(**args)
|
return AsyncEngineArgs(**args)
|
||||||
|
|||||||
+4
-2
@@ -1,10 +1,12 @@
|
|||||||
from transformers import AutoTokenizer
|
import logging
|
||||||
import os
|
import os
|
||||||
from typing import Union
|
from typing import Union
|
||||||
|
|
||||||
|
from transformers import AutoTokenizer
|
||||||
|
|
||||||
class TokenizerWrapper:
|
class TokenizerWrapper:
|
||||||
def __init__(self, tokenizer_name_or_path, tokenizer_revision, trust_remote_code):
|
def __init__(self, tokenizer_name_or_path, tokenizer_revision, trust_remote_code):
|
||||||
print(f"tokenizer_name_or_path: {tokenizer_name_or_path}, tokenizer_revision: {tokenizer_revision}, trust_remote_code: {trust_remote_code}")
|
logging.debug("tokenizer_name_or_path: %s, tokenizer_revision: %s, trust_remote_code: %s", tokenizer_name_or_path, tokenizer_revision, trust_remote_code)
|
||||||
self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_name_or_path, revision=tokenizer_revision or "main", trust_remote_code=trust_remote_code)
|
self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_name_or_path, revision=tokenizer_revision or "main", trust_remote_code=trust_remote_code)
|
||||||
self.custom_chat_template = os.getenv("CUSTOM_CHAT_TEMPLATE")
|
self.custom_chat_template = os.getenv("CUSTOM_CHAT_TEMPLATE")
|
||||||
self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template)
|
self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template)
|
||||||
|
|||||||
Reference in New Issue
Block a user