Compare commits

...
19 Commits
Author SHA1 Message Date
chrisvelaandGitHub 1b3228a2dc Merge pull request #307 from runpod-workers/fix/revert-0.20.0
Release / release (push) Waiting to run
revert to 0.20.2
2026-06-12 15:50:52 -05:00
velaraptor-runpod 4817d4a8e7 revert to 0.20.2 2026-06-12 15:49:23 -05:00
chrisvelaandGitHub 0378382a92 Merge pull request #306 from runpod-workers/revert/v2.20.1
Release / release (push) Waiting to run
chore: carry non-vllm changes from main (tests GPU + configs)
2026-06-12 15:21:22 -05:00
velaraptor-runpodandClaude Sonnet 4.6 08580e7ccf chore: carry non-vllm changes from main (tests GPU + configs)
Brings forward the L40 GPU type in tests.json and the new llama/qwen
tuned config files, while keeping Dockerfile pinned at vllm 0.20.2
(v2.20.1 state).

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-12 15:18:36 -05:00
chrisvelaandGitHub 8868aae6b1 Merge pull request #305 from runpod-workers/fix/typo-tests-l40
Release / release (push) Waiting to run
fix: fix typo in tests
2026-06-12 14:00:54 -05:00
velaraptor-runpod fb8adc5c06 fix: fix typo in tests 2026-06-12 13:59:29 -05:00
chrisvelaandGitHub 7351da512b Merge pull request #304 from runpod-workers/fix/fix-tests
Release / release (push) Waiting to run
fix: change test gpu to L40
2026-06-12 13:49:15 -05:00
velaraptor-runpod 1e78043b2c fix: change test gpu to L40 2026-06-12 13:46:33 -05:00
chrisvelaandGitHub 352c64f4c1 Merge pull request #303 from runpod-workers/chore/check-vllm-versions-readmes
chore: check readmes vllm version in sync with version in Dockerfile
2026-06-12 09:49:07 -05:00
velaraptor-runpod 0922f5b435 chore: check readmes vllm version in sync with version in Dockerfile 2026-06-11 17:33:16 -05:00
chrisvelaandGitHub 5d9a48fc70 Merge pull request #302 from runpod-workers/feat/0.22.1
Release / release (push) Waiting to run
feat: upgrade vllm to 0.22.1
2026-06-11 17:27:38 -05:00
velaraptor-runpod 5d1579e361 fix: update readme links 2026-06-11 17:00:31 -05:00
velaraptor-runpod 0488b77d89 feat: upgrade vllm to 0.22.1 2026-06-11 16:29:02 -05:00
chrisvelaandGitHub d3a962c33b Merge pull request #301 from adithyaJRunpod/feature/tuned-configs
Release / release (push) Waiting to run
Add tuned configs,  CON-239
2026-06-11 14:26:00 -05:00
chrisvelaandGitHub 105c125698 Merge pull request #300 from runpod-workers/feat/0.21.0
feat: upgrade vllm to 0.21.0
2026-06-11 12:47:08 -05:00
AdithyaJob 0a0ccfcb60 Add tuned configs for Llama 3.1 8B and Qwen3 8B 2026-06-10 21:07:41 -07:00
velaraptor-runpod c8ce53c72c fix: add kenels, and fix for cuda 2026-06-10 16:01:53 -05:00
velaraptor-runpod 9618e799ba chore: fix cuda libraries 2026-06-04 15:35:58 -05:00
velaraptor-runpod cb3f077dba feat: upgrade vllm to 0.21.0 2026-06-03 14:43:56 -05:00
6 changed files with 94 additions and 3 deletions
@@ -0,0 +1,71 @@
name: CI | Sync vLLM version in READMEs
on:
push:
branches: ["main"]
paths:
- "Dockerfile"
workflow_dispatch:
permissions:
contents: write
pull-requests: write
jobs:
sync_version:
runs-on: ubuntu-latest
name: Check README version matches Dockerfile and update if needed
steps:
- name: Checkout
uses: actions/checkout@v4
- name: Extract vLLM version from Dockerfile and sync READMEs
run: |
echo "Extracting vLLM version from Dockerfile..."
dockerfile_version=$(grep -oP 'vllm(?:\[[\w,]+\])?==\K[\d.]+' Dockerfile | head -1)
if [ -z "$dockerfile_version" ]; then
echo "ERROR: Could not extract vLLM version from Dockerfile."
exit 1
fi
echo "Dockerfile vLLM version: $dockerfile_version"
echo "VLLM_VERSION=$dockerfile_version" >> $GITHUB_ENV
updated=0
for readme in README.md .runpod/README.md; do
if [ ! -f "$readme" ]; then
echo "Skipping $readme (not found)"
continue
fi
readme_version=$(grep -oP 'Current vLLM version: \[\K[\d.]+' "$readme" || echo "")
echo "$readme current version: ${readme_version:-not found}"
if [ "$readme_version" = "$dockerfile_version" ]; then
echo "$readme is already up to date."
continue
fi
echo "Updating $readme from $readme_version to $dockerfile_version..."
sed -i "s|Current vLLM version: \[${readme_version}\](https://github.com/vllm-project/vllm/releases/tag/v${readme_version})|Current vLLM version: [${dockerfile_version}](https://github.com/vllm-project/vllm/releases/tag/v${dockerfile_version})|g" "$readme"
updated=1
done
echo "UPDATED=$updated" >> $GITHUB_ENV
- name: Create Pull Request
if: env.UPDATED == '1'
uses: peter-evans/create-pull-request@v7
with:
token: ${{ secrets.GITHUB_TOKEN }}
commit-message: "docs: sync vLLM version to ${{ env.VLLM_VERSION }} in READMEs"
title: "docs: sync vLLM version to ${{ env.VLLM_VERSION }} in READMEs"
body: |
The vLLM version in the Dockerfile has been updated to `${{ env.VLLM_VERSION }}`.
This PR syncs the version badge/link in:
- `README.md`
- `.runpod/README.md`
branch: docs/sync-vllm-version-${{ env.VLLM_VERSION }}
labels: documentation
+1 -1
View File
@@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
[![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
Current vLLM version: [0.22.1](https://github.com/vllm-project/vllm/releases/tag/v0.22.1)
---
+1 -1
View File
@@ -30,7 +30,7 @@
}
],
"config": {
"gpuTypeId": "NVIDIA GeForce RTX 4090",
"gpuTypeId": "NVIDIA L40",
"gpuCount": 1,
"env": [
{
+1 -1
View File
@@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png)
Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
Current vLLM version: [0.22.1](https://github.com/vllm-project/vllm/releases/tag/v0.22.1)
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
+10
View File
@@ -0,0 +1,10 @@
model: meta-llama/Llama-3.1-8B-Instruct
gpu-memory-utilization: 0.95
max-model-len: 8192
dtype: auto
trust-remote-code: true
quantization: fp8
kv-cache-dtype: fp8
enforce-eager: false
enable-prefix-caching: true
speculative-config: '{"model":"RedHatAI/Llama-3.1-8B-Instruct-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
+10
View File
@@ -0,0 +1,10 @@
model: Qwen/Qwen3-8B
gpu-memory-utilization: 0.95
max-model-len: 8192
dtype: auto
trust-remote-code: true
quantization: fp8
kv-cache-dtype: fp8
enforce-eager: false
enable-prefix-caching: true
speculative-config: '{"model":"RedHatAI/Qwen3-8B-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'