From 9de17d49b72e9a4e6c465968fb3c9a9cb48cffbf Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 30 Apr 2026 17:25:23 -0500 Subject: [PATCH 01/13] feat: update vllm to 0.18.1 --- Dockerfile | 2 +- vllm-loadbalancer-ep.code-workspace | 11 +++++++++++ 2 files changed, 12 insertions(+), 1 deletion(-) create mode 100644 vllm-loadbalancer-ep.code-workspace diff --git a/Dockerfile b/Dockerfile index e59a03e..332bf48 100644 --- a/Dockerfile +++ b/Dockerfile @@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/ # Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.17.1" --extra-index-url https://download.pytorch.org/whl/cu129 + uv pip install --system "vllm[flashinfer]==0.18.1" --extra-index-url https://download.pytorch.org/whl/cu129 # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) COPY builder/requirements.txt /requirements.txt diff --git a/vllm-loadbalancer-ep.code-workspace b/vllm-loadbalancer-ep.code-workspace new file mode 100644 index 0000000..c29e4b2 --- /dev/null +++ b/vllm-loadbalancer-ep.code-workspace @@ -0,0 +1,11 @@ +{ + "folders": [ + { + "path": "../vllm-loadbalancer-ep" + }, + { + "path": "." + } + ], + "settings": {} +} \ No newline at end of file From a774cefe854de2cc88393967e3167adaac5742a2 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 30 Apr 2026 17:29:00 -0500 Subject: [PATCH 02/13] fix: clean workspace file --- vllm-loadbalancer-ep.code-workspace | 11 ----------- 1 file changed, 11 deletions(-) delete mode 100644 vllm-loadbalancer-ep.code-workspace diff --git a/vllm-loadbalancer-ep.code-workspace b/vllm-loadbalancer-ep.code-workspace deleted file mode 100644 index c29e4b2..0000000 --- a/vllm-loadbalancer-ep.code-workspace +++ /dev/null @@ -1,11 +0,0 @@ -{ - "folders": [ - { - "path": "../vllm-loadbalancer-ep" - }, - { - "path": "." - } - ], - "settings": {} -} \ No newline at end of file From e6950bdebdc81a03eaf877cabba9076e725ed3fe Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 30 Apr 2026 17:41:11 -0500 Subject: [PATCH 03/13] feat: upgrade vLLM to 0.19.1 - Bump vllm[flashinfer] to 0.19.1 in Dockerfile - Add OpenAIServingRender (new required dependency in 0.19.x serving layer) - Pass openai_serving_render to all four serving class constructors - Remove log_error_stack param (removed upstream in 0.19.x) Co-Authored-By: Claude Sonnet 4.6 --- Dockerfile | 2 +- src/engine.py | 30 ++++++++++++++++++++++++------ 2 files changed, 25 insertions(+), 7 deletions(-) diff --git a/Dockerfile b/Dockerfile index 332bf48..1f05b31 100644 --- a/Dockerfile +++ b/Dockerfile @@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/ # Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.18.1" --extra-index-url https://download.pytorch.org/whl/cu129 + uv pip install --system "vllm[flashinfer]==0.19.1" --extra-index-url https://download.pytorch.org/whl/cu129 # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) COPY builder/requirements.txt /requirements.txt diff --git a/src/engine.py b/src/engine.py index 8c99326..dcbf67a 100644 --- a/src/engine.py +++ b/src/engine.py @@ -19,6 +19,7 @@ from vllm.entrypoints.openai.models.protocol import BaseModelPath, LoRAModulePat from vllm.entrypoints.openai.models.serving import OpenAIServingModels from vllm.entrypoints.openai.responses.protocol import ResponsesRequest, ResponsesResponse from vllm.entrypoints.openai.responses.serving import OpenAIServingResponses +from vllm.entrypoints.serve.render.serving import OpenAIServingRender from constants import DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MAX_CONCURRENCY, DEFAULT_MIN_BATCH_SIZE from engine_args import get_engine_args @@ -246,16 +247,33 @@ class OpenAIvLLMEngine(vLLMEngine): lora_modules=self.lora_adapters, ) await self.serving_models.init_static_loras() - + # Get chat template from vLLM tokenizer if available chat_template = None if self.tokenizer and hasattr(self.tokenizer, 'tokenizer'): chat_template = self.tokenizer.tokenizer.chat_template - + + self.openai_serving_render = OpenAIServingRender( + model_config=self.llm.model_config, + renderer=self.llm.renderer, + io_processor=self.llm.io_processor, + model_registry=self.serving_models.registry, + request_logger=None, + chat_template=chat_template, + chat_template_content_format="auto", + trust_request_chat_template=os.getenv('TRUST_REQUEST_CHAT_TEMPLATE', 'false').lower() == 'true', + enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true', + exclude_tools_when_tool_choice_none=os.getenv('EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE', 'false').lower() == 'true', + tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None, + reasoning_parser=os.getenv('REASONING_PARSER', "") or None, + log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true', + ) + self.chat_engine = OpenAIServingChat( - engine_client=self.llm, + engine_client=self.llm, models=self.serving_models, response_role=self.response_role, + openai_serving_render=self.openai_serving_render, request_logger=None, chat_template=chat_template, chat_template_content_format="auto", @@ -268,20 +286,20 @@ class OpenAIvLLMEngine(vLLMEngine): enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true', enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true', enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true', - log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true', ) self.completion_engine = OpenAIServingCompletion( engine_client=self.llm, models=self.serving_models, + openai_serving_render=self.openai_serving_render, request_logger=None, return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true', enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true', enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true', - log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true', ) self.responses_engine = OpenAIServingResponses( engine_client=self.llm, models=self.serving_models, + openai_serving_render=self.openai_serving_render, request_logger=None, chat_template=chat_template, chat_template_content_format="auto", @@ -293,12 +311,12 @@ class OpenAIvLLMEngine(vLLMEngine): enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true', enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true', enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true', - log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true', ) self.messages_engine = AnthropicServingMessages( engine_client=self.llm, models=self.serving_models, response_role=self.response_role, + openai_serving_render=self.openai_serving_render, request_logger=None, chat_template=chat_template, chat_template_content_format="auto", From 178c72238e7b5bf91895a9917b947216f8be410a Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 30 Apr 2026 17:49:09 -0500 Subject: [PATCH 04/13] fix: surface LORA_MODULES parse failures instead of silently loading zero adapters Fixes FDE-194. Previously a malformed LORA_MODULES value was swallowed at info level and the engine would start with no LoRA adapters, causing 500s on any request using an adapter model name (e.g. npc-sim-*). Changes: - Log at error level when LORA_MODULES cannot be parsed as JSON - Log at error level when individual adapter dicts fail LoRAModulePath validation - Log a final error when all adapters fail to load so the cause is obvious - Accept a single adapter dict (not just an array) for convenience - Return early when LORA_MODULES is unset to skip unnecessary parsing Co-Authored-By: Claude Sonnet 4.6 --- src/engine.py | 49 +++++++++++++++++++++++++++++++++++++++---------- 1 file changed, 39 insertions(+), 10 deletions(-) diff --git a/src/engine.py b/src/engine.py index dcbf67a..80bf72b 100644 --- a/src/engine.py +++ b/src/engine.py @@ -206,19 +206,48 @@ class OpenAIvLLMEngine(vLLMEngine): self.raw_openai_output = bool(int(raw_output_env)) def _load_lora_adapters(self): - adapters = [] - try: - adapters = json.loads(os.getenv("LORA_MODULES", '[]')) - except Exception as e: - logging.info(f"---Initialized adapter json load error: {e}") + lora_modules_env = os.getenv("LORA_MODULES", "") + if not lora_modules_env: + return [] - for i, adapter in enumerate(adapters): + try: + parsed = json.loads(lora_modules_env) + except json.JSONDecodeError as e: + logging.error( + "LORA_MODULES could not be parsed as JSON: %s — no LoRA adapters loaded. Value: %r", + e, lora_modules_env, + ) + return [] + + # Accept a single adapter dict as well as an array + if isinstance(parsed, dict): + parsed = [parsed] + + if not isinstance(parsed, list): + logging.error( + "LORA_MODULES must be a JSON array of adapter objects, got %s — no LoRA adapters loaded.", + type(parsed).__name__, + ) + return [] + + adapters = [] + for i, adapter in enumerate(parsed): try: - adapters[i] = LoRAModulePath(**adapter) - logging.info(f"---Initialized adapter: {adapter}") + adapters.append(LoRAModulePath(**adapter)) + logging.info("Loaded LoRA adapter config [%d]: %s", i, adapter) except Exception as e: - logging.info(f"---Initialized adapter not worked: {e}") - continue + logging.error( + "Failed to parse LoRA adapter at index %d: %s. Config: %r", + i, e, adapter, + ) + + if parsed and not adapters: + logging.error( + "LORA_MODULES specified %d adapter(s) but none could be loaded — " + "OpenAI model name lookups for LoRA adapters will fail.", + len(parsed), + ) + return adapters async def _ensure_engines_initialized(self): From fa42ecd79a51701f27b1864d314eda72ddd200f8 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 30 Apr 2026 17:54:14 -0500 Subject: [PATCH 05/13] fix: resolve lowercase HF cache paths when MODEL_NAME uses original casing Fixes FDE-174. Some model stores (e.g. RunPod pre-cached network volumes) normalize repo IDs to lowercase. HuggingFace Hub caches using the original casing, so MODEL_NAME=Qwen/Qwen2.5-Coder-32B-Instruct-AWQ would miss a cache stored as models--qwen--qwen2.5-coder-32b-instruct-awq/ and attempt a redundant download that fails on limited container storage. If the exact-case HF cache directory is absent but a lowercase variant exists, the latest snapshot path is returned directly so vLLM loads from disk. Absolute paths and models with no lowercase cache are unchanged. Co-Authored-By: Claude Sonnet 4.6 --- src/engine_args.py | 56 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 56 insertions(+) diff --git a/src/engine_args.py b/src/engine_args.py index 8ed132c..3ba1844 100644 --- a/src/engine_args.py +++ b/src/engine_args.py @@ -342,6 +342,58 @@ def _sanitize_hf_overrides(hf_overrides: dict) -> dict | None: return result or None +def _resolve_cached_model_path(model_name: str) -> str: + """Return a local snapshot path when the HF cache was stored with lowercase names. + + Some model stores (e.g. RunPod pre-cached volumes) normalize repo IDs to + lowercase. HuggingFace Hub stores caches as + ``models--{org}--{model}/snapshots/{hash}/`` preserving the original casing, + so MODEL_NAME=Qwen/Qwen2.5-Coder-32B-Instruct-AWQ will miss a cache stored + as ``models--qwen--qwen2.5-coder-32b-instruct-awq/``. + + If the exact-case cache directory is absent but a lowercase variant exists, + the latest snapshot path is returned so vLLM loads from disk rather than + attempting a redundant download. + """ + if os.path.isabs(model_name): + return model_name + + cache_dir = ( + os.getenv("HUGGINGFACE_HUB_CACHE") + or os.getenv("HF_HOME") + or os.path.expanduser("~/.cache/huggingface/hub") + ) + + folder_name = f"models--{model_name.replace('/', '--')}" + + if os.path.isdir(os.path.join(cache_dir, folder_name)): + return model_name + + lower_dir = os.path.join(cache_dir, folder_name.lower()) + if not os.path.isdir(lower_dir): + return model_name + + snapshots_dir = os.path.join(lower_dir, "snapshots") + if not os.path.isdir(snapshots_dir): + return model_name + + try: + snapshots = sorted(os.listdir(snapshots_dir)) + except OSError: + return model_name + + if not snapshots: + return model_name + + resolved = os.path.join(snapshots_dir, snapshots[-1]) + logging.info( + "MODEL_NAME %r not found at original casing in HF cache; " + "resolved to lowercase cached snapshot at %r", + model_name, resolved, + ) + return resolved + + def get_local_args(): """ Retrieve local arguments from a JSON file. @@ -517,4 +569,8 @@ def get_engine_args(): if speculative_config: args["speculative_config"] = speculative_config + # Resolve lowercase HF cache paths (FDE-174) + if args.get("model"): + args["model"] = _resolve_cached_model_path(args["model"]) + return AsyncEngineArgs(**args) From 4f8a16df5db3dec3bb13889a311725cc1388a33e Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 30 Apr 2026 19:41:09 -0500 Subject: [PATCH 06/13] fix: update transformers to >=5 --- builder/requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/builder/requirements.txt b/builder/requirements.txt index 6cc3dae..c7bc129 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -9,7 +9,7 @@ typing-extensions>=4.8.0 pydantic pydantic-settings hf-transfer -transformers>=4.57.0,<5 +transformers>=5 bitsandbytes>=0.45.0 kernels torch-c-dlpack-ext From 577fd8c3c3afc5b26338d0313484b42516284a8f Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 30 Apr 2026 20:22:24 -0500 Subject: [PATCH 07/13] fix: fix old github actions, trigger release on publish release --- .github/workflows/CI-runpod_dep.yml | 63 +++++++++++++++-------------- .github/workflows/release.yml | 13 +++--- .runpod/README.md | 2 + README.md | 2 +- 4 files changed, 41 insertions(+), 39 deletions(-) diff --git a/.github/workflows/CI-runpod_dep.yml b/.github/workflows/CI-runpod_dep.yml index e99d828..f77f0c9 100644 --- a/.github/workflows/CI-runpod_dep.yml +++ b/.github/workflows/CI-runpod_dep.yml @@ -9,59 +9,60 @@ on: workflow_dispatch: +permissions: + contents: write + pull-requests: write + jobs: check_dep: runs-on: ubuntu-latest name: Check python requirements file and update steps: - name: Checkout - uses: actions/checkout@v2 + uses: actions/checkout@v4 - name: Check for new package version and update run: | - echo "Fetching the current runpod version from requirements.txt..." - - # Get current version, allowing both == and ~= in the search pattern - current_version=$(grep -oP 'runpod[~=]{1,2}\K[^"]+' ./builder/requirements.txt) - echo "Current version: $current_version" + echo "Fetching current runpod version from requirements.txt..." - # Extract major and minor from current version - current_major_minor=$(echo $current_version | cut -d. -f1,2) - echo "Current major.minor: $current_major_minor" + # Match runpod with any version specifier or no specifier at all + current_version=$(grep -oP '^runpod([~>=!<]{1,2}\K[\d.]+)?' ./builder/requirements.txt | grep -oP '[\d.]+' || echo "") + echo "Current version: ${current_version:-unset}" - echo "Fetching the latest runpod version from PyPI..." - - # Get new version from PyPI - new_version=$(curl -s https://pypi.org/pypi/runpod/json | jq -r .info.version) + echo "Fetching latest runpod version from PyPI..." + new_version=$(curl -sf https://pypi.org/pypi/runpod/json | jq -r .info.version) echo "NEW_VERSION_ENV=$new_version" >> $GITHUB_ENV echo "New version: $new_version" - # Extract major and minor from new version - new_major_minor=$(echo $new_version | cut -d. -f1,2) - echo "New major.minor: $new_major_minor" - if [ -z "$new_version" ]; then - echo "ERROR: Failed to fetch the new version from PyPI." - exit 1 + echo "ERROR: Failed to fetch new version from PyPI." + exit 1 fi - # Check if the major or minor version is different - if [ "$current_major_minor" = "$new_major_minor" ]; then - echo "No update needed. The new version ($new_major_minor) is within the allowed range (~= $current_major_minor)." + if [ -z "$current_version" ]; then + echo "No version pin found — pinning to $new_version." + else + current_major_minor=$(echo "$current_version" | cut -d. -f1,2) + new_major_minor=$(echo "$new_version" | cut -d. -f1,2) + echo "Current major.minor: $current_major_minor New major.minor: $new_major_minor" + + if [ "$current_major_minor" = "$new_major_minor" ]; then + echo "No update needed. New version ($new_version) is within ~= $current_major_minor range." exit 0 + fi + + echo "New major/minor detected ($new_major_minor). Updating requirements.txt..." fi - echo "New major/minor detected ($new_major_minor). Updating requirements.txt..." - - # Update requirements.txt, preserving the existing constraint type (~= or ==) - sed -i "s/runpod[~=][^ ]*/runpod~=$new_version/" ./builder/requirements.txt - echo "requirements.txt has been updated." + # Replace any `runpod`, `runpod==x`, `runpod~=x`, etc. with pinned version + sed -i "s|^runpod.*|runpod~=$new_version|" ./builder/requirements.txt + echo "requirements.txt updated." - name: Create Pull Request - uses: peter-evans/create-pull-request@v3 + uses: peter-evans/create-pull-request@v7 with: token: ${{ secrets.GITHUB_TOKEN }} - commit-message: Update runpod package version - title: Update runpod package version - body: The package version has been updated to ${{ env.NEW_VERSION_ENV }} + commit-message: "chore: update runpod to ${{ env.NEW_VERSION_ENV }}" + title: "chore: update runpod to ${{ env.NEW_VERSION_ENV }}" + body: The `runpod` package has been updated to `${{ env.NEW_VERSION_ENV }}`. branch: runpod-package-update diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 3fd1e60..8a2afcd 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -1,9 +1,11 @@ name: Release on: + release: + types: [published] push: tags: - - "v[0-9]+.[0-9]+.[0-9]+*" # Trigger on version tags like v1.0.0, v2.1.0, etc. + - "v[0-9]+.[0-9]+.[0-9]+*" workflow_dispatch: inputs: version: @@ -53,16 +55,13 @@ jobs: # Determine version based on trigger type if [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then - # Manual trigger: use input version VERSION="${{ github.event.inputs.version }}" - echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV - echo "IS_MANUAL_RELEASE=true" >> $GITHUB_ENV + elif [[ "${{ github.event_name }}" == "release" ]]; then + VERSION="${{ github.event.release.tag_name }}" else - # Tag trigger: use tag name (remove refs/tags/ prefix) VERSION=${GITHUB_REF#refs/tags/} - echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV - echo "IS_MANUAL_RELEASE=false" >> $GITHUB_ENV fi + echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV - name: Build and push the images to Docker Hub uses: docker/bake-action@v2 diff --git a/.runpod/README.md b/.runpod/README.md index 6eb8652..b4739ea 100644 --- a/.runpod/README.md +++ b/.runpod/README.md @@ -6,6 +6,8 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API [![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) +Current vLLM version: [0.17.1](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) + --- ## Endpoint Configuration diff --git a/README.md b/README.md index 0c7f0e5..9dd8742 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https: ![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png) -Current vLLM version: [0.16.0](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) +Current vLLM version: [0.17.1](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) > Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep) From 72547aa3bb27a9fb946435525353258556f151bb Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 30 Apr 2026 20:25:02 -0500 Subject: [PATCH 08/13] chore: update readme vllm version --- .runpod/README.md | 1 + README.md | 2 +- 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/.runpod/README.md b/.runpod/README.md index 6eb8652..c28a2ce 100644 --- a/.runpod/README.md +++ b/.runpod/README.md @@ -6,6 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API [![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) +Current vLLM version: [0.18.1](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) --- ## Endpoint Configuration diff --git a/README.md b/README.md index 0c7f0e5..9914c5f 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https: ![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png) -Current vLLM version: [0.16.0](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) +Current vLLM version: [0.18.1](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) > Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep) From 3d4af5df9bb91389374ce94ef56eab842a92db11 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 30 Apr 2026 20:25:35 -0500 Subject: [PATCH 09/13] chore: update readme vllm version --- .runpod/README.md | 2 ++ README.md | 2 +- 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/.runpod/README.md b/.runpod/README.md index 6eb8652..bdc1878 100644 --- a/.runpod/README.md +++ b/.runpod/README.md @@ -6,6 +6,8 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API [![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) +Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) + --- ## Endpoint Configuration diff --git a/README.md b/README.md index 0c7f0e5..136f099 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https: ![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png) -Current vLLM version: [0.16.0](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) +Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) > Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep) From 0cb8aeae77dc38ea2e57a0444e15d6a4fce151d7 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Fri, 1 May 2026 09:47:23 -0500 Subject: [PATCH 10/13] chore: add specific runpod version --- builder/requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/builder/requirements.txt b/builder/requirements.txt index 6cc3dae..b4f9072 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -1,7 +1,7 @@ ray pandas pyarrow -runpod +runpod==1.9.0 huggingface-hub lmcache==0.4.2 packaging>=24.2 From 0140b29c445c1a46547eecf6799c8c70e6a6c401 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Fri, 1 May 2026 09:48:43 -0500 Subject: [PATCH 11/13] chore: add release notes to slack notification --- .github/workflows/release.yml | 42 ++++++++++++++++++++++++++--------- 1 file changed, 32 insertions(+), 10 deletions(-) diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 8a2afcd..33e3f20 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -79,19 +79,41 @@ jobs: echo "Version: ${{ env.RELEASE_VERSION }}" echo "Docker Image: ${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}" + - name: Fetch Release Notes + run: | + RESPONSE=$(curl -sf \ + -H "Authorization: token ${{ github.token }}" \ + "https://api.github.com/repos/${{ github.repository }}/releases/tags/${{ env.RELEASE_VERSION }}" 2>/dev/null) || true + if [[ -n "$RESPONSE" ]]; then + NOTES=$(echo "$RESPONSE" | jq -r '.body // empty') + fi + printf '%s' "${NOTES:-No release notes available.}" > /tmp/release_notes.txt + - name: Notify Slack run: | - curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \ - -H "Content-Type: application/json" \ - -d '{ - "text": ":rocket: New :runpod-new-whiteonpurple: Runpod worker-vllm release: *${{ env.RELEASE_VERSION }}*", - "blocks": [ + jq -n \ + --arg version "${{ env.RELEASE_VERSION }}" \ + --arg docker "${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}" \ + --rawfile notes /tmp/release_notes.txt \ + --arg url "https://github.com/${{ github.repository }}/releases/tag/${{ env.RELEASE_VERSION }}" \ + '{ + text: (":rocket: New :runpod-new-whiteonpurple: Runpod worker-vllm release: *" + $version + "*"), + blocks: [ { - "type": "section", - "text": { - "type": "mrkdwn", - "text": ":banana-dance: *New Release — worker-vllm ${{ env.RELEASE_VERSION }}*\n*Docker:* `${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}`\n" + type: "section", + text: { + type: "mrkdwn", + text: (":banana-dance: *New Release — worker-vllm " + $version + "*\n*Docker:* `" + $docker + "`\n<" + $url + "|View release on GitHub>") + } + }, + { + type: "section", + text: { + type: "mrkdwn", + text: ("*Release Notes:*\n" + $notes) } } ] - }' + }' | curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \ + -H "Content-Type: application/json" \ + -d @- From 7dc853b1fea4cde5d17ad1694bdcf6845750981d Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Fri, 1 May 2026 10:13:32 -0500 Subject: [PATCH 12/13] chore: remove release to trigger on release publish, just use tags. duplicate --- .github/workflows/release.yml | 2 -- 1 file changed, 2 deletions(-) diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 33e3f20..18eeab8 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -1,8 +1,6 @@ name: Release on: - release: - types: [published] push: tags: - "v[0-9]+.[0-9]+.[0-9]+*" From ff87840a58d586f96c6f282c33ad41f21f9003f1 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Fri, 1 May 2026 14:13:03 -0500 Subject: [PATCH 13/13] fix: add enforce_eager as true, add pytorch_alloc_conf to expandle_segments to True for OOM, for hub defaults --- .runpod/README.md | 1 + .runpod/hub.json | 12 +++++++++++- 2 files changed, 12 insertions(+), 1 deletion(-) diff --git a/.runpod/README.md b/.runpod/README.md index 3c80029..d93c258 100644 --- a/.runpod/README.md +++ b/.runpod/README.md @@ -29,6 +29,7 @@ All behaviour is controlled through environment variables: | `REASONING_PARSER` | Parser for reasoning-capable models | | "deepseek_r1", "qwen3", "granite", "hunyuan_a13b" | | `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | Override served model name in API | | String | | `MAX_CONCURRENCY` | Maximum concurrent requests | 300 | Integer | +| `ENFORCE_EAGER` | If True, we will disable CUDA graph and always execute the model in eager mode. If False, we will use CUDA graph and eager execution in hybrid for maximal performance and flexibility. | true | boolean (true or false) | **Pass any vLLM engine arg** not listed above by setting an env var with the **UPPERCASED** field name (e.g. `MAX_MODEL_LEN=4096`, `ENABLE_CHUNKED_PREFILL=true`). The worker auto-discovers all `AsyncEngineArgs` fields from env. See the [vLLM engine args docs](https://docs.vllm.ai/en/latest/configuration/engine_args) for all available options. diff --git a/.runpod/hub.json b/.runpod/hub.json index a45aa55..6ab87fb 100644 --- a/.runpod/hub.json +++ b/.runpod/hub.json @@ -621,7 +621,7 @@ "name": "Enforce Eager", "type": "boolean", "description": "Always use eager-mode PyTorch. If False (0), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility", - "default": false, + "default": true, "advanced": true } }, @@ -795,6 +795,16 @@ "default": "", "advanced": true } + }, + { + "key": "PYTORCH_ALLOC_CONF", + "input": { + "name": "PyTorch Alloc Config", + "type": "string", + "description": "PyTorch allocation configuration, remove this if you want to use the default configuration", + "default": "expandable_segments:True", + "advanced": true + } } ] }