Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8868aae6b1 | ||
|
|
fb8adc5c06 | ||
|
|
7351da512b | ||
|
|
1e78043b2c | ||
|
|
352c64f4c1 | ||
|
|
0922f5b435 | ||
|
|
5d9a48fc70 | ||
|
|
5d1579e361 | ||
|
|
0488b77d89 | ||
|
|
d3a962c33b | ||
|
|
105c125698 | ||
|
|
0a0ccfcb60 | ||
|
|
c8ce53c72c | ||
|
|
9618e799ba | ||
|
|
cb3f077dba |
@@ -0,0 +1,71 @@
|
|||||||
|
name: CI | Sync vLLM version in READMEs
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches: ["main"]
|
||||||
|
paths:
|
||||||
|
- "Dockerfile"
|
||||||
|
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: write
|
||||||
|
pull-requests: write
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
sync_version:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
name: Check README version matches Dockerfile and update if needed
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Extract vLLM version from Dockerfile and sync READMEs
|
||||||
|
run: |
|
||||||
|
echo "Extracting vLLM version from Dockerfile..."
|
||||||
|
dockerfile_version=$(grep -oP 'vllm(?:\[[\w,]+\])?==\K[\d.]+' Dockerfile | head -1)
|
||||||
|
|
||||||
|
if [ -z "$dockerfile_version" ]; then
|
||||||
|
echo "ERROR: Could not extract vLLM version from Dockerfile."
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
echo "Dockerfile vLLM version: $dockerfile_version"
|
||||||
|
echo "VLLM_VERSION=$dockerfile_version" >> $GITHUB_ENV
|
||||||
|
|
||||||
|
updated=0
|
||||||
|
for readme in README.md .runpod/README.md; do
|
||||||
|
if [ ! -f "$readme" ]; then
|
||||||
|
echo "Skipping $readme (not found)"
|
||||||
|
continue
|
||||||
|
fi
|
||||||
|
|
||||||
|
readme_version=$(grep -oP 'Current vLLM version: \[\K[\d.]+' "$readme" || echo "")
|
||||||
|
echo "$readme current version: ${readme_version:-not found}"
|
||||||
|
|
||||||
|
if [ "$readme_version" = "$dockerfile_version" ]; then
|
||||||
|
echo "$readme is already up to date."
|
||||||
|
continue
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "Updating $readme from $readme_version to $dockerfile_version..."
|
||||||
|
sed -i "s|Current vLLM version: \[${readme_version}\](https://github.com/vllm-project/vllm/releases/tag/v${readme_version})|Current vLLM version: [${dockerfile_version}](https://github.com/vllm-project/vllm/releases/tag/v${dockerfile_version})|g" "$readme"
|
||||||
|
updated=1
|
||||||
|
done
|
||||||
|
|
||||||
|
echo "UPDATED=$updated" >> $GITHUB_ENV
|
||||||
|
|
||||||
|
- name: Create Pull Request
|
||||||
|
if: env.UPDATED == '1'
|
||||||
|
uses: peter-evans/create-pull-request@v7
|
||||||
|
with:
|
||||||
|
token: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
commit-message: "docs: sync vLLM version to ${{ env.VLLM_VERSION }} in READMEs"
|
||||||
|
title: "docs: sync vLLM version to ${{ env.VLLM_VERSION }} in READMEs"
|
||||||
|
body: |
|
||||||
|
The vLLM version in the Dockerfile has been updated to `${{ env.VLLM_VERSION }}`.
|
||||||
|
|
||||||
|
This PR syncs the version badge/link in:
|
||||||
|
- `README.md`
|
||||||
|
- `.runpod/README.md`
|
||||||
|
branch: docs/sync-vllm-version-${{ env.VLLM_VERSION }}
|
||||||
|
labels: documentation
|
||||||
+1
-1
@@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
|
|||||||
|
|
||||||
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
||||||
|
|
||||||
Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
|
Current vLLM version: [0.22.1](https://github.com/vllm-project/vllm/releases/tag/v0.22.1)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -30,7 +30,7 @@
|
|||||||
}
|
}
|
||||||
],
|
],
|
||||||
"config": {
|
"config": {
|
||||||
"gpuTypeId": "NVIDIA GeForce RTX 4090",
|
"gpuTypeId": "NVIDIA L40",
|
||||||
"gpuCount": 1,
|
"gpuCount": 1,
|
||||||
"env": [
|
"env": [
|
||||||
{
|
{
|
||||||
|
|||||||
+12
-1
@@ -8,9 +8,20 @@ ENV PATH="/root/.local/bin:$PATH"
|
|||||||
|
|
||||||
RUN ldconfig /usr/local/cuda-13.0/compat/
|
RUN ldconfig /usr/local/cuda-13.0/compat/
|
||||||
|
|
||||||
|
# nixl_ep PyPI wheels are compiled against CUDA 12.x and require libcudart.so.12.
|
||||||
|
# CUDA 13 runtime is ABI-compatible with CUDA 12, so symlinking is safe.
|
||||||
|
# Symlink into /usr/local/cuda/lib64 (already in LD_LIBRARY_PATH) so the linker
|
||||||
|
# finds it by filename scan rather than relying on ldcache SONAME lookup.
|
||||||
|
RUN ln -sf /usr/local/cuda/lib64/libcudart.so.13 /usr/local/cuda/lib64/libcudart.so.12 && ldconfig
|
||||||
|
|
||||||
|
# CUDA 13.0 containers return libs to /usr/local/nvidia/lib64 so container
|
||||||
|
# providers (RunPod, Lambda, etc.) can mount host drivers there consistently.
|
||||||
|
# See: https://github.com/vllm-project/vllm/issues/18859
|
||||||
|
ENV LD_LIBRARY_PATH=/usr/local/nvidia/lib64:/usr/local/cuda/lib64:$LD_LIBRARY_PATH
|
||||||
|
|
||||||
# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
|
# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
|
||||||
RUN uv pip install --system "packaging>=24.2" && \
|
RUN uv pip install --system "packaging>=24.2" && \
|
||||||
uv pip install --system "vllm[flashinfer]==0.20.2" && \
|
uv pip install --system "vllm[flashinfer]==0.22.1" && \
|
||||||
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation
|
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation
|
||||||
|
|
||||||
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
|
|||||||
|
|
||||||

|

|
||||||
|
|
||||||
Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
|
Current vLLM version: [0.22.1](https://github.com/vllm-project/vllm/releases/tag/v0.22.1)
|
||||||
|
|
||||||
|
|
||||||
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
|
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
|
||||||
|
|||||||
@@ -3,7 +3,7 @@ pandas
|
|||||||
pyarrow
|
pyarrow
|
||||||
runpod==1.9.1
|
runpod==1.9.1
|
||||||
huggingface-hub
|
huggingface-hub
|
||||||
lmcache==0.4.5
|
lmcache==0.4.6
|
||||||
packaging>=24.2
|
packaging>=24.2
|
||||||
typing-extensions>=4.8.0
|
typing-extensions>=4.8.0
|
||||||
pydantic
|
pydantic
|
||||||
|
|||||||
@@ -0,0 +1,10 @@
|
|||||||
|
model: meta-llama/Llama-3.1-8B-Instruct
|
||||||
|
gpu-memory-utilization: 0.95
|
||||||
|
max-model-len: 8192
|
||||||
|
dtype: auto
|
||||||
|
trust-remote-code: true
|
||||||
|
quantization: fp8
|
||||||
|
kv-cache-dtype: fp8
|
||||||
|
enforce-eager: false
|
||||||
|
enable-prefix-caching: true
|
||||||
|
speculative-config: '{"model":"RedHatAI/Llama-3.1-8B-Instruct-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||||
@@ -0,0 +1,10 @@
|
|||||||
|
model: Qwen/Qwen3-8B
|
||||||
|
gpu-memory-utilization: 0.95
|
||||||
|
max-model-len: 8192
|
||||||
|
dtype: auto
|
||||||
|
trust-remote-code: true
|
||||||
|
quantization: fp8
|
||||||
|
kv-cache-dtype: fp8
|
||||||
|
enforce-eager: false
|
||||||
|
enable-prefix-caching: true
|
||||||
|
speculative-config: '{"model":"RedHatAI/Qwen3-8B-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||||
Reference in New Issue
Block a user