diff --git a/.github/workflows/CI-runpod_dep.yml b/.github/workflows/CI-runpod_dep.yml index e99d828..f77f0c9 100644 --- a/.github/workflows/CI-runpod_dep.yml +++ b/.github/workflows/CI-runpod_dep.yml @@ -9,59 +9,60 @@ on: workflow_dispatch: +permissions: + contents: write + pull-requests: write + jobs: check_dep: runs-on: ubuntu-latest name: Check python requirements file and update steps: - name: Checkout - uses: actions/checkout@v2 + uses: actions/checkout@v4 - name: Check for new package version and update run: | - echo "Fetching the current runpod version from requirements.txt..." - - # Get current version, allowing both == and ~= in the search pattern - current_version=$(grep -oP 'runpod[~=]{1,2}\K[^"]+' ./builder/requirements.txt) - echo "Current version: $current_version" + echo "Fetching current runpod version from requirements.txt..." - # Extract major and minor from current version - current_major_minor=$(echo $current_version | cut -d. -f1,2) - echo "Current major.minor: $current_major_minor" + # Match runpod with any version specifier or no specifier at all + current_version=$(grep -oP '^runpod([~>=!<]{1,2}\K[\d.]+)?' ./builder/requirements.txt | grep -oP '[\d.]+' || echo "") + echo "Current version: ${current_version:-unset}" - echo "Fetching the latest runpod version from PyPI..." - - # Get new version from PyPI - new_version=$(curl -s https://pypi.org/pypi/runpod/json | jq -r .info.version) + echo "Fetching latest runpod version from PyPI..." + new_version=$(curl -sf https://pypi.org/pypi/runpod/json | jq -r .info.version) echo "NEW_VERSION_ENV=$new_version" >> $GITHUB_ENV echo "New version: $new_version" - # Extract major and minor from new version - new_major_minor=$(echo $new_version | cut -d. -f1,2) - echo "New major.minor: $new_major_minor" - if [ -z "$new_version" ]; then - echo "ERROR: Failed to fetch the new version from PyPI." - exit 1 + echo "ERROR: Failed to fetch new version from PyPI." + exit 1 fi - # Check if the major or minor version is different - if [ "$current_major_minor" = "$new_major_minor" ]; then - echo "No update needed. The new version ($new_major_minor) is within the allowed range (~= $current_major_minor)." + if [ -z "$current_version" ]; then + echo "No version pin found — pinning to $new_version." + else + current_major_minor=$(echo "$current_version" | cut -d. -f1,2) + new_major_minor=$(echo "$new_version" | cut -d. -f1,2) + echo "Current major.minor: $current_major_minor New major.minor: $new_major_minor" + + if [ "$current_major_minor" = "$new_major_minor" ]; then + echo "No update needed. New version ($new_version) is within ~= $current_major_minor range." exit 0 + fi + + echo "New major/minor detected ($new_major_minor). Updating requirements.txt..." fi - echo "New major/minor detected ($new_major_minor). Updating requirements.txt..." - - # Update requirements.txt, preserving the existing constraint type (~= or ==) - sed -i "s/runpod[~=][^ ]*/runpod~=$new_version/" ./builder/requirements.txt - echo "requirements.txt has been updated." + # Replace any `runpod`, `runpod==x`, `runpod~=x`, etc. with pinned version + sed -i "s|^runpod.*|runpod~=$new_version|" ./builder/requirements.txt + echo "requirements.txt updated." - name: Create Pull Request - uses: peter-evans/create-pull-request@v3 + uses: peter-evans/create-pull-request@v7 with: token: ${{ secrets.GITHUB_TOKEN }} - commit-message: Update runpod package version - title: Update runpod package version - body: The package version has been updated to ${{ env.NEW_VERSION_ENV }} + commit-message: "chore: update runpod to ${{ env.NEW_VERSION_ENV }}" + title: "chore: update runpod to ${{ env.NEW_VERSION_ENV }}" + body: The `runpod` package has been updated to `${{ env.NEW_VERSION_ENV }}`. branch: runpod-package-update diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 3fd1e60..18eeab8 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -3,7 +3,7 @@ name: Release on: push: tags: - - "v[0-9]+.[0-9]+.[0-9]+*" # Trigger on version tags like v1.0.0, v2.1.0, etc. + - "v[0-9]+.[0-9]+.[0-9]+*" workflow_dispatch: inputs: version: @@ -53,16 +53,13 @@ jobs: # Determine version based on trigger type if [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then - # Manual trigger: use input version VERSION="${{ github.event.inputs.version }}" - echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV - echo "IS_MANUAL_RELEASE=true" >> $GITHUB_ENV + elif [[ "${{ github.event_name }}" == "release" ]]; then + VERSION="${{ github.event.release.tag_name }}" else - # Tag trigger: use tag name (remove refs/tags/ prefix) VERSION=${GITHUB_REF#refs/tags/} - echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV - echo "IS_MANUAL_RELEASE=false" >> $GITHUB_ENV fi + echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV - name: Build and push the images to Docker Hub uses: docker/bake-action@v2 @@ -80,19 +77,41 @@ jobs: echo "Version: ${{ env.RELEASE_VERSION }}" echo "Docker Image: ${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}" + - name: Fetch Release Notes + run: | + RESPONSE=$(curl -sf \ + -H "Authorization: token ${{ github.token }}" \ + "https://api.github.com/repos/${{ github.repository }}/releases/tags/${{ env.RELEASE_VERSION }}" 2>/dev/null) || true + if [[ -n "$RESPONSE" ]]; then + NOTES=$(echo "$RESPONSE" | jq -r '.body // empty') + fi + printf '%s' "${NOTES:-No release notes available.}" > /tmp/release_notes.txt + - name: Notify Slack run: | - curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \ - -H "Content-Type: application/json" \ - -d '{ - "text": ":rocket: New :runpod-new-whiteonpurple: Runpod worker-vllm release: *${{ env.RELEASE_VERSION }}*", - "blocks": [ + jq -n \ + --arg version "${{ env.RELEASE_VERSION }}" \ + --arg docker "${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}" \ + --rawfile notes /tmp/release_notes.txt \ + --arg url "https://github.com/${{ github.repository }}/releases/tag/${{ env.RELEASE_VERSION }}" \ + '{ + text: (":rocket: New :runpod-new-whiteonpurple: Runpod worker-vllm release: *" + $version + "*"), + blocks: [ { - "type": "section", - "text": { - "type": "mrkdwn", - "text": ":banana-dance: *New Release — worker-vllm ${{ env.RELEASE_VERSION }}*\n*Docker:* `${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}`\n" + type: "section", + text: { + type: "mrkdwn", + text: (":banana-dance: *New Release — worker-vllm " + $version + "*\n*Docker:* `" + $docker + "`\n<" + $url + "|View release on GitHub>") + } + }, + { + type: "section", + text: { + type: "mrkdwn", + text: ("*Release Notes:*\n" + $notes) } } ] - }' + }' | curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \ + -H "Content-Type: application/json" \ + -d @- diff --git a/.runpod/README.md b/.runpod/README.md index 6eb8652..d93c258 100644 --- a/.runpod/README.md +++ b/.runpod/README.md @@ -6,6 +6,8 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API [![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) +Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.19.1) + --- ## Endpoint Configuration @@ -27,6 +29,7 @@ All behaviour is controlled through environment variables: | `REASONING_PARSER` | Parser for reasoning-capable models | | "deepseek_r1", "qwen3", "granite", "hunyuan_a13b" | | `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | Override served model name in API | | String | | `MAX_CONCURRENCY` | Maximum concurrent requests | 300 | Integer | +| `ENFORCE_EAGER` | If True, we will disable CUDA graph and always execute the model in eager mode. If False, we will use CUDA graph and eager execution in hybrid for maximal performance and flexibility. | true | boolean (true or false) | **Pass any vLLM engine arg** not listed above by setting an env var with the **UPPERCASED** field name (e.g. `MAX_MODEL_LEN=4096`, `ENABLE_CHUNKED_PREFILL=true`). The worker auto-discovers all `AsyncEngineArgs` fields from env. See the [vLLM engine args docs](https://docs.vllm.ai/en/latest/configuration/engine_args) for all available options. diff --git a/.runpod/hub.json b/.runpod/hub.json index a45aa55..6ab87fb 100644 --- a/.runpod/hub.json +++ b/.runpod/hub.json @@ -621,7 +621,7 @@ "name": "Enforce Eager", "type": "boolean", "description": "Always use eager-mode PyTorch. If False (0), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility", - "default": false, + "default": true, "advanced": true } }, @@ -795,6 +795,16 @@ "default": "", "advanced": true } + }, + { + "key": "PYTORCH_ALLOC_CONF", + "input": { + "name": "PyTorch Alloc Config", + "type": "string", + "description": "PyTorch allocation configuration, remove this if you want to use the default configuration", + "default": "expandable_segments:True", + "advanced": true + } } ] } diff --git a/Dockerfile b/Dockerfile index e59a03e..1f05b31 100644 --- a/Dockerfile +++ b/Dockerfile @@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/ # Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.17.1" --extra-index-url https://download.pytorch.org/whl/cu129 + uv pip install --system "vllm[flashinfer]==0.19.1" --extra-index-url https://download.pytorch.org/whl/cu129 # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) COPY builder/requirements.txt /requirements.txt diff --git a/README.md b/README.md index 0c7f0e5..7d18213 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,8 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https: ![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png) -Current vLLM version: [0.16.0](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) +Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.19.1) + > Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep) diff --git a/builder/requirements.txt b/builder/requirements.txt index 6cc3dae..f3ad976 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -1,7 +1,7 @@ ray pandas pyarrow -runpod +runpod==1.9.0 huggingface-hub lmcache==0.4.2 packaging>=24.2 @@ -9,7 +9,7 @@ typing-extensions>=4.8.0 pydantic pydantic-settings hf-transfer -transformers>=4.57.0,<5 +transformers>=5 bitsandbytes>=0.45.0 kernels torch-c-dlpack-ext diff --git a/src/engine.py b/src/engine.py index 8c99326..80bf72b 100644 --- a/src/engine.py +++ b/src/engine.py @@ -19,6 +19,7 @@ from vllm.entrypoints.openai.models.protocol import BaseModelPath, LoRAModulePat from vllm.entrypoints.openai.models.serving import OpenAIServingModels from vllm.entrypoints.openai.responses.protocol import ResponsesRequest, ResponsesResponse from vllm.entrypoints.openai.responses.serving import OpenAIServingResponses +from vllm.entrypoints.serve.render.serving import OpenAIServingRender from constants import DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MAX_CONCURRENCY, DEFAULT_MIN_BATCH_SIZE from engine_args import get_engine_args @@ -205,19 +206,48 @@ class OpenAIvLLMEngine(vLLMEngine): self.raw_openai_output = bool(int(raw_output_env)) def _load_lora_adapters(self): - adapters = [] - try: - adapters = json.loads(os.getenv("LORA_MODULES", '[]')) - except Exception as e: - logging.info(f"---Initialized adapter json load error: {e}") + lora_modules_env = os.getenv("LORA_MODULES", "") + if not lora_modules_env: + return [] - for i, adapter in enumerate(adapters): + try: + parsed = json.loads(lora_modules_env) + except json.JSONDecodeError as e: + logging.error( + "LORA_MODULES could not be parsed as JSON: %s — no LoRA adapters loaded. Value: %r", + e, lora_modules_env, + ) + return [] + + # Accept a single adapter dict as well as an array + if isinstance(parsed, dict): + parsed = [parsed] + + if not isinstance(parsed, list): + logging.error( + "LORA_MODULES must be a JSON array of adapter objects, got %s — no LoRA adapters loaded.", + type(parsed).__name__, + ) + return [] + + adapters = [] + for i, adapter in enumerate(parsed): try: - adapters[i] = LoRAModulePath(**adapter) - logging.info(f"---Initialized adapter: {adapter}") + adapters.append(LoRAModulePath(**adapter)) + logging.info("Loaded LoRA adapter config [%d]: %s", i, adapter) except Exception as e: - logging.info(f"---Initialized adapter not worked: {e}") - continue + logging.error( + "Failed to parse LoRA adapter at index %d: %s. Config: %r", + i, e, adapter, + ) + + if parsed and not adapters: + logging.error( + "LORA_MODULES specified %d adapter(s) but none could be loaded — " + "OpenAI model name lookups for LoRA adapters will fail.", + len(parsed), + ) + return adapters async def _ensure_engines_initialized(self): @@ -246,16 +276,33 @@ class OpenAIvLLMEngine(vLLMEngine): lora_modules=self.lora_adapters, ) await self.serving_models.init_static_loras() - + # Get chat template from vLLM tokenizer if available chat_template = None if self.tokenizer and hasattr(self.tokenizer, 'tokenizer'): chat_template = self.tokenizer.tokenizer.chat_template - + + self.openai_serving_render = OpenAIServingRender( + model_config=self.llm.model_config, + renderer=self.llm.renderer, + io_processor=self.llm.io_processor, + model_registry=self.serving_models.registry, + request_logger=None, + chat_template=chat_template, + chat_template_content_format="auto", + trust_request_chat_template=os.getenv('TRUST_REQUEST_CHAT_TEMPLATE', 'false').lower() == 'true', + enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true', + exclude_tools_when_tool_choice_none=os.getenv('EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE', 'false').lower() == 'true', + tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None, + reasoning_parser=os.getenv('REASONING_PARSER', "") or None, + log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true', + ) + self.chat_engine = OpenAIServingChat( - engine_client=self.llm, + engine_client=self.llm, models=self.serving_models, response_role=self.response_role, + openai_serving_render=self.openai_serving_render, request_logger=None, chat_template=chat_template, chat_template_content_format="auto", @@ -268,20 +315,20 @@ class OpenAIvLLMEngine(vLLMEngine): enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true', enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true', enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true', - log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true', ) self.completion_engine = OpenAIServingCompletion( engine_client=self.llm, models=self.serving_models, + openai_serving_render=self.openai_serving_render, request_logger=None, return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true', enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true', enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true', - log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true', ) self.responses_engine = OpenAIServingResponses( engine_client=self.llm, models=self.serving_models, + openai_serving_render=self.openai_serving_render, request_logger=None, chat_template=chat_template, chat_template_content_format="auto", @@ -293,12 +340,12 @@ class OpenAIvLLMEngine(vLLMEngine): enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true', enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true', enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true', - log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true', ) self.messages_engine = AnthropicServingMessages( engine_client=self.llm, models=self.serving_models, response_role=self.response_role, + openai_serving_render=self.openai_serving_render, request_logger=None, chat_template=chat_template, chat_template_content_format="auto", diff --git a/src/engine_args.py b/src/engine_args.py index f6d92d9..f801f6e 100644 --- a/src/engine_args.py +++ b/src/engine_args.py @@ -352,6 +352,58 @@ def _sanitize_hf_overrides(hf_overrides: dict) -> dict | None: return result or None +def _resolve_cached_model_path(model_name: str) -> str: + """Return a local snapshot path when the HF cache was stored with lowercase names. + + Some model stores (e.g. RunPod pre-cached volumes) normalize repo IDs to + lowercase. HuggingFace Hub stores caches as + ``models--{org}--{model}/snapshots/{hash}/`` preserving the original casing, + so MODEL_NAME=Qwen/Qwen2.5-Coder-32B-Instruct-AWQ will miss a cache stored + as ``models--qwen--qwen2.5-coder-32b-instruct-awq/``. + + If the exact-case cache directory is absent but a lowercase variant exists, + the latest snapshot path is returned so vLLM loads from disk rather than + attempting a redundant download. + """ + if os.path.isabs(model_name): + return model_name + + cache_dir = ( + os.getenv("HUGGINGFACE_HUB_CACHE") + or os.getenv("HF_HOME") + or os.path.expanduser("~/.cache/huggingface/hub") + ) + + folder_name = f"models--{model_name.replace('/', '--')}" + + if os.path.isdir(os.path.join(cache_dir, folder_name)): + return model_name + + lower_dir = os.path.join(cache_dir, folder_name.lower()) + if not os.path.isdir(lower_dir): + return model_name + + snapshots_dir = os.path.join(lower_dir, "snapshots") + if not os.path.isdir(snapshots_dir): + return model_name + + try: + snapshots = sorted(os.listdir(snapshots_dir)) + except OSError: + return model_name + + if not snapshots: + return model_name + + resolved = os.path.join(snapshots_dir, snapshots[-1]) + logging.info( + "MODEL_NAME %r not found at original casing in HF cache; " + "resolved to lowercase cached snapshot at %r", + model_name, resolved, + ) + return resolved + + def get_local_args(): """ Retrieve local arguments from a JSON file. @@ -527,4 +579,8 @@ def get_engine_args(): if speculative_config: args["speculative_config"] = speculative_config + # Resolve lowercase HF cache paths (FDE-174) + if args.get("model"): + args["model"] = _resolve_cached_model_path(args["model"]) + return AsyncEngineArgs(**args)