diff --git a/.github/workflows/CI-runpod_dep.yml b/.github/workflows/CI-runpod_dep.yml index e99d828..f77f0c9 100644 --- a/.github/workflows/CI-runpod_dep.yml +++ b/.github/workflows/CI-runpod_dep.yml @@ -9,59 +9,60 @@ on: workflow_dispatch: +permissions: + contents: write + pull-requests: write + jobs: check_dep: runs-on: ubuntu-latest name: Check python requirements file and update steps: - name: Checkout - uses: actions/checkout@v2 + uses: actions/checkout@v4 - name: Check for new package version and update run: | - echo "Fetching the current runpod version from requirements.txt..." - - # Get current version, allowing both == and ~= in the search pattern - current_version=$(grep -oP 'runpod[~=]{1,2}\K[^"]+' ./builder/requirements.txt) - echo "Current version: $current_version" + echo "Fetching current runpod version from requirements.txt..." - # Extract major and minor from current version - current_major_minor=$(echo $current_version | cut -d. -f1,2) - echo "Current major.minor: $current_major_minor" + # Match runpod with any version specifier or no specifier at all + current_version=$(grep -oP '^runpod([~>=!<]{1,2}\K[\d.]+)?' ./builder/requirements.txt | grep -oP '[\d.]+' || echo "") + echo "Current version: ${current_version:-unset}" - echo "Fetching the latest runpod version from PyPI..." - - # Get new version from PyPI - new_version=$(curl -s https://pypi.org/pypi/runpod/json | jq -r .info.version) + echo "Fetching latest runpod version from PyPI..." + new_version=$(curl -sf https://pypi.org/pypi/runpod/json | jq -r .info.version) echo "NEW_VERSION_ENV=$new_version" >> $GITHUB_ENV echo "New version: $new_version" - # Extract major and minor from new version - new_major_minor=$(echo $new_version | cut -d. -f1,2) - echo "New major.minor: $new_major_minor" - if [ -z "$new_version" ]; then - echo "ERROR: Failed to fetch the new version from PyPI." - exit 1 + echo "ERROR: Failed to fetch new version from PyPI." + exit 1 fi - # Check if the major or minor version is different - if [ "$current_major_minor" = "$new_major_minor" ]; then - echo "No update needed. The new version ($new_major_minor) is within the allowed range (~= $current_major_minor)." + if [ -z "$current_version" ]; then + echo "No version pin found — pinning to $new_version." + else + current_major_minor=$(echo "$current_version" | cut -d. -f1,2) + new_major_minor=$(echo "$new_version" | cut -d. -f1,2) + echo "Current major.minor: $current_major_minor New major.minor: $new_major_minor" + + if [ "$current_major_minor" = "$new_major_minor" ]; then + echo "No update needed. New version ($new_version) is within ~= $current_major_minor range." exit 0 + fi + + echo "New major/minor detected ($new_major_minor). Updating requirements.txt..." fi - echo "New major/minor detected ($new_major_minor). Updating requirements.txt..." - - # Update requirements.txt, preserving the existing constraint type (~= or ==) - sed -i "s/runpod[~=][^ ]*/runpod~=$new_version/" ./builder/requirements.txt - echo "requirements.txt has been updated." + # Replace any `runpod`, `runpod==x`, `runpod~=x`, etc. with pinned version + sed -i "s|^runpod.*|runpod~=$new_version|" ./builder/requirements.txt + echo "requirements.txt updated." - name: Create Pull Request - uses: peter-evans/create-pull-request@v3 + uses: peter-evans/create-pull-request@v7 with: token: ${{ secrets.GITHUB_TOKEN }} - commit-message: Update runpod package version - title: Update runpod package version - body: The package version has been updated to ${{ env.NEW_VERSION_ENV }} + commit-message: "chore: update runpod to ${{ env.NEW_VERSION_ENV }}" + title: "chore: update runpod to ${{ env.NEW_VERSION_ENV }}" + body: The `runpod` package has been updated to `${{ env.NEW_VERSION_ENV }}`. branch: runpod-package-update diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 37a55c9..18eeab8 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -3,7 +3,7 @@ name: Release on: push: tags: - - "v[0-9]+.[0-9]+.[0-9]+*" # Trigger on version tags like v1.0.0, v2.1.0, etc. + - "v[0-9]+.[0-9]+.[0-9]+*" workflow_dispatch: inputs: version: @@ -53,16 +53,13 @@ jobs: # Determine version based on trigger type if [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then - # Manual trigger: use input version VERSION="${{ github.event.inputs.version }}" - echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV - echo "IS_MANUAL_RELEASE=true" >> $GITHUB_ENV + elif [[ "${{ github.event_name }}" == "release" ]]; then + VERSION="${{ github.event.release.tag_name }}" else - # Tag trigger: use tag name (remove refs/tags/ prefix) VERSION=${GITHUB_REF#refs/tags/} - echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV - echo "IS_MANUAL_RELEASE=false" >> $GITHUB_ENV fi + echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV - name: Build and push the images to Docker Hub uses: docker/bake-action@v2 @@ -76,11 +73,45 @@ jobs: - name: Release Summary run: | - echo "🚀 Release completed!" + echo "Release completed!" echo "Version: ${{ env.RELEASE_VERSION }}" echo "Docker Image: ${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}" - if [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then - echo "Trigger: Manual workflow dispatch" - else - echo "Trigger: GitHub release (tag: ${{ github.ref_name }})" + + - name: Fetch Release Notes + run: | + RESPONSE=$(curl -sf \ + -H "Authorization: token ${{ github.token }}" \ + "https://api.github.com/repos/${{ github.repository }}/releases/tags/${{ env.RELEASE_VERSION }}" 2>/dev/null) || true + if [[ -n "$RESPONSE" ]]; then + NOTES=$(echo "$RESPONSE" | jq -r '.body // empty') fi + printf '%s' "${NOTES:-No release notes available.}" > /tmp/release_notes.txt + + - name: Notify Slack + run: | + jq -n \ + --arg version "${{ env.RELEASE_VERSION }}" \ + --arg docker "${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}" \ + --rawfile notes /tmp/release_notes.txt \ + --arg url "https://github.com/${{ github.repository }}/releases/tag/${{ env.RELEASE_VERSION }}" \ + '{ + text: (":rocket: New :runpod-new-whiteonpurple: Runpod worker-vllm release: *" + $version + "*"), + blocks: [ + { + type: "section", + text: { + type: "mrkdwn", + text: (":banana-dance: *New Release — worker-vllm " + $version + "*\n*Docker:* `" + $docker + "`\n<" + $url + "|View release on GitHub>") + } + }, + { + type: "section", + text: { + type: "mrkdwn", + text: ("*Release Notes:*\n" + $notes) + } + } + ] + }' | curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \ + -H "Content-Type: application/json" \ + -d @- diff --git a/.github/workflows/slack-pr-issue-notify.yml b/.github/workflows/slack-pr-issue-notify.yml new file mode 100644 index 0000000..539f5df --- /dev/null +++ b/.github/workflows/slack-pr-issue-notify.yml @@ -0,0 +1,41 @@ +name: Slack PR Notifications + +on: + pull_request: + types: [opened] + issues: + types: [opened] + +permissions: + contents: read + +jobs: + notify: + runs-on: ubuntu-latest + steps: + - name: Notify Slack - New PR + if: github.event_name == 'pull_request' + run: | + curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \ + -H "Content-Type: application/json" \ + -d '{ + "text": ":rocket: New PR in worker-vllm: *${{ github.event.pull_request.title }}*", + "blocks": [ + { + "type": "section", + "text": { + "type": "mrkdwn", + "text": ":rocket: *New Pull Request — worker-vllm*\n*<${{ github.event.pull_request.html_url }}|${{ github.event.pull_request.title }}>*\nOpened by *${{ github.event.pull_request.user.login }}*" + } + }, + { + "type": "context", + "elements": [ + { + "type": "mrkdwn", + "text": "${{ github.event.pull_request.base.ref }} ← ${{ github.event.pull_request.head.ref }}" + } + ] + } + ] + }' diff --git a/.github/workflows/slack-vllm-monitor.yml b/.github/workflows/slack-vllm-monitor.yml new file mode 100644 index 0000000..2a57b47 --- /dev/null +++ b/.github/workflows/slack-vllm-monitor.yml @@ -0,0 +1,73 @@ +name: Monitor vLLM Releases + +on: + schedule: + - cron: '0 0 * * *' # Every day at midnight + workflow_dispatch: + + +permissions: + contents: read + +jobs: + check-vllm-release: + runs-on: ubuntu-latest + steps: + - name: Restore last known vLLM tag + uses: actions/cache/restore@v4 + with: + path: .vllm-last-tag + key: vllm-tag-${{ github.run_id }} + restore-keys: vllm-tag- + + - name: Get latest vLLM release + id: vllm + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + response=$(curl -sf https://api.github.com/repos/vllm-project/vllm/releases/latest \ + -H "Authorization: Bearer $GH_TOKEN") + echo "tag=$(echo "$response" | jq -r '.tag_name')" >> $GITHUB_OUTPUT + echo "url=$(echo "$response" | jq -r '.html_url')" >> $GITHUB_OUTPUT + echo "name=$(echo "$response" | jq -r '.name')" >> $GITHUB_OUTPUT + + - name: Check if new release + id: check + run: | + last=$(cat .vllm-last-tag 2>/dev/null || echo "") + current="${{ steps.vllm.outputs.tag }}" + echo "Last: $last Current: $current" + if [ -n "$current" ] && [ "$last" != "$current" ]; then + echo "is_new=true" >> $GITHUB_OUTPUT + else + echo "is_new=false" >> $GITHUB_OUTPUT + fi + + - name: Notify Slack + if: steps.check.outputs.is_new == 'true' + run: | + curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \ + -H "Content-Type: application/json" \ + -d '{ + "text": ":rocket: New vLLM release: *${{ steps.vllm.outputs.tag }}*", + "blocks": [ + { + "type": "section", + "text": { + "type": "mrkdwn", + "text": ":rocket: *New vLLM Release: ${{ steps.vllm.outputs.tag }}*\n<${{ steps.vllm.outputs.url }}|View on GitHub>" + } + } + ] + }' + + - name: Save new tag + if: steps.check.outputs.is_new == 'true' + run: echo "${{ steps.vllm.outputs.tag }}" > .vllm-last-tag + + - name: Update cache + if: steps.check.outputs.is_new == 'true' + uses: actions/cache/save@v4 + with: + path: .vllm-last-tag + key: vllm-tag-${{ steps.vllm.outputs.tag }} diff --git a/.runpod/README.md b/.runpod/README.md index b813ab8..18e4369 100644 --- a/.runpod/README.md +++ b/.runpod/README.md @@ -6,8 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API [![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) -Current vLLM version: [0.20.0](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) - +Current vLLM version: [0.20.0](https://github.com/vllm-project/vllm/releases/tag/v0.20.0) --- @@ -30,6 +29,7 @@ All behaviour is controlled through environment variables: | `REASONING_PARSER` | Parser for reasoning-capable models | | "deepseek_r1", "qwen3", "granite", "hunyuan_a13b" | | `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | Override served model name in API | | String | | `MAX_CONCURRENCY` | Maximum concurrent requests | 300 | Integer | +| `ENFORCE_EAGER` | If True, we will disable CUDA graph and always execute the model in eager mode. If False, we will use CUDA graph and eager execution in hybrid for maximal performance and flexibility. | true | boolean (true or false) | **Pass any vLLM engine arg** not listed above by setting an env var with the **UPPERCASED** field name (e.g. `MAX_MODEL_LEN=4096`, `ENABLE_CHUNKED_PREFILL=true`). The worker auto-discovers all `AsyncEngineArgs` fields from env. See the [vLLM engine args docs](https://docs.vllm.ai/en/latest/configuration/engine_args) for all available options. diff --git a/.runpod/hub.json b/.runpod/hub.json index a45aa55..6ab87fb 100644 --- a/.runpod/hub.json +++ b/.runpod/hub.json @@ -621,7 +621,7 @@ "name": "Enforce Eager", "type": "boolean", "description": "Always use eager-mode PyTorch. If False (0), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility", - "default": false, + "default": true, "advanced": true } }, @@ -795,6 +795,16 @@ "default": "", "advanced": true } + }, + { + "key": "PYTORCH_ALLOC_CONF", + "input": { + "name": "PyTorch Alloc Config", + "type": "string", + "description": "PyTorch allocation configuration, remove this if you want to use the default configuration", + "default": "expandable_segments:True", + "advanced": true + } } ] } diff --git a/README.md b/README.md index 7426fcf..42dd66d 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,8 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https: ![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png) -Current vLLM version: [0.20.0](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) +Current vLLM version: [0.20.0](https://github.com/vllm-project/vllm/releases/tag/v0.20.0) + > Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep) diff --git a/builder/requirements.txt b/builder/requirements.txt index 6cc3dae..f3ad976 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -1,7 +1,7 @@ ray pandas pyarrow -runpod +runpod==1.9.0 huggingface-hub lmcache==0.4.2 packaging>=24.2 @@ -9,7 +9,7 @@ typing-extensions>=4.8.0 pydantic pydantic-settings hf-transfer -transformers>=4.57.0,<5 +transformers>=5 bitsandbytes>=0.45.0 kernels torch-c-dlpack-ext