Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
87d7365126 | ||
|
|
0e83616f93 | ||
|
|
ab6d39dcf8 | ||
|
|
73f030ae5e | ||
|
|
8a099c1723 | ||
|
|
ff87840a58 | ||
|
|
7dc853b1fe | ||
|
|
a8b754b92a | ||
|
|
6357aeda51 | ||
|
|
0140b29c44 | ||
|
|
0cb8aeae77 | ||
|
|
cd8f9e9560 | ||
|
|
895fd25fac | ||
|
|
7bb8df73af | ||
|
|
3d4af5df9b | ||
|
|
72547aa3bb | ||
|
|
577fd8c3c3 | ||
|
|
4f8a16df5d | ||
|
|
fa42ecd79a | ||
|
|
178c72238e | ||
|
|
e6950bdebd | ||
|
|
a774cefe85 | ||
|
|
9de17d49b7 |
@@ -9,59 +9,60 @@ on:
|
||||
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
|
||||
jobs:
|
||||
check_dep:
|
||||
runs-on: ubuntu-latest
|
||||
name: Check python requirements file and update
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v2
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Check for new package version and update
|
||||
run: |
|
||||
echo "Fetching the current runpod version from requirements.txt..."
|
||||
echo "Fetching current runpod version from requirements.txt..."
|
||||
|
||||
# Get current version, allowing both == and ~= in the search pattern
|
||||
current_version=$(grep -oP 'runpod[~=]{1,2}\K[^"]+' ./builder/requirements.txt)
|
||||
echo "Current version: $current_version"
|
||||
# Match runpod with any version specifier or no specifier at all
|
||||
current_version=$(grep -oP '^runpod([~>=!<]{1,2}\K[\d.]+)?' ./builder/requirements.txt | grep -oP '[\d.]+' || echo "")
|
||||
echo "Current version: ${current_version:-unset}"
|
||||
|
||||
# Extract major and minor from current version
|
||||
current_major_minor=$(echo $current_version | cut -d. -f1,2)
|
||||
echo "Current major.minor: $current_major_minor"
|
||||
|
||||
echo "Fetching the latest runpod version from PyPI..."
|
||||
|
||||
# Get new version from PyPI
|
||||
new_version=$(curl -s https://pypi.org/pypi/runpod/json | jq -r .info.version)
|
||||
echo "Fetching latest runpod version from PyPI..."
|
||||
new_version=$(curl -sf https://pypi.org/pypi/runpod/json | jq -r .info.version)
|
||||
echo "NEW_VERSION_ENV=$new_version" >> $GITHUB_ENV
|
||||
echo "New version: $new_version"
|
||||
|
||||
# Extract major and minor from new version
|
||||
new_major_minor=$(echo $new_version | cut -d. -f1,2)
|
||||
echo "New major.minor: $new_major_minor"
|
||||
|
||||
if [ -z "$new_version" ]; then
|
||||
echo "ERROR: Failed to fetch the new version from PyPI."
|
||||
echo "ERROR: Failed to fetch new version from PyPI."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Check if the major or minor version is different
|
||||
if [ -z "$current_version" ]; then
|
||||
echo "No version pin found — pinning to $new_version."
|
||||
else
|
||||
current_major_minor=$(echo "$current_version" | cut -d. -f1,2)
|
||||
new_major_minor=$(echo "$new_version" | cut -d. -f1,2)
|
||||
echo "Current major.minor: $current_major_minor New major.minor: $new_major_minor"
|
||||
|
||||
if [ "$current_major_minor" = "$new_major_minor" ]; then
|
||||
echo "No update needed. The new version ($new_major_minor) is within the allowed range (~= $current_major_minor)."
|
||||
echo "No update needed. New version ($new_version) is within ~= $current_major_minor range."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "New major/minor detected ($new_major_minor). Updating requirements.txt..."
|
||||
fi
|
||||
|
||||
# Update requirements.txt, preserving the existing constraint type (~= or ==)
|
||||
sed -i "s/runpod[~=][^ ]*/runpod~=$new_version/" ./builder/requirements.txt
|
||||
echo "requirements.txt has been updated."
|
||||
# Replace any `runpod`, `runpod==x`, `runpod~=x`, etc. with pinned version
|
||||
sed -i "s|^runpod.*|runpod~=$new_version|" ./builder/requirements.txt
|
||||
echo "requirements.txt updated."
|
||||
|
||||
- name: Create Pull Request
|
||||
uses: peter-evans/create-pull-request@v3
|
||||
uses: peter-evans/create-pull-request@v7
|
||||
with:
|
||||
token: ${{ secrets.GITHUB_TOKEN }}
|
||||
commit-message: Update runpod package version
|
||||
title: Update runpod package version
|
||||
body: The package version has been updated to ${{ env.NEW_VERSION_ENV }}
|
||||
commit-message: "chore: update runpod to ${{ env.NEW_VERSION_ENV }}"
|
||||
title: "chore: update runpod to ${{ env.NEW_VERSION_ENV }}"
|
||||
body: The `runpod` package has been updated to `${{ env.NEW_VERSION_ENV }}`.
|
||||
branch: runpod-package-update
|
||||
|
||||
@@ -3,7 +3,7 @@ name: Release
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- "v[0-9]+.[0-9]+.[0-9]+*" # Trigger on version tags like v1.0.0, v2.1.0, etc.
|
||||
- "v[0-9]+.[0-9]+.[0-9]+*"
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
version:
|
||||
@@ -53,16 +53,13 @@ jobs:
|
||||
|
||||
# Determine version based on trigger type
|
||||
if [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then
|
||||
# Manual trigger: use input version
|
||||
VERSION="${{ github.event.inputs.version }}"
|
||||
echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV
|
||||
echo "IS_MANUAL_RELEASE=true" >> $GITHUB_ENV
|
||||
elif [[ "${{ github.event_name }}" == "release" ]]; then
|
||||
VERSION="${{ github.event.release.tag_name }}"
|
||||
else
|
||||
# Tag trigger: use tag name (remove refs/tags/ prefix)
|
||||
VERSION=${GITHUB_REF#refs/tags/}
|
||||
echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV
|
||||
echo "IS_MANUAL_RELEASE=false" >> $GITHUB_ENV
|
||||
fi
|
||||
echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV
|
||||
|
||||
- name: Build and push the images to Docker Hub
|
||||
uses: docker/bake-action@v2
|
||||
@@ -80,19 +77,41 @@ jobs:
|
||||
echo "Version: ${{ env.RELEASE_VERSION }}"
|
||||
echo "Docker Image: ${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}"
|
||||
|
||||
- name: Fetch Release Notes
|
||||
run: |
|
||||
RESPONSE=$(curl -sf \
|
||||
-H "Authorization: token ${{ github.token }}" \
|
||||
"https://api.github.com/repos/${{ github.repository }}/releases/tags/${{ env.RELEASE_VERSION }}" 2>/dev/null) || true
|
||||
if [[ -n "$RESPONSE" ]]; then
|
||||
NOTES=$(echo "$RESPONSE" | jq -r '.body // empty')
|
||||
fi
|
||||
printf '%s' "${NOTES:-No release notes available.}" > /tmp/release_notes.txt
|
||||
|
||||
- name: Notify Slack
|
||||
run: |
|
||||
curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d '{
|
||||
"text": ":rocket: New :runpod-new-whiteonpurple: Runpod worker-vllm release: *${{ env.RELEASE_VERSION }}*",
|
||||
"blocks": [
|
||||
jq -n \
|
||||
--arg version "${{ env.RELEASE_VERSION }}" \
|
||||
--arg docker "${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}" \
|
||||
--rawfile notes /tmp/release_notes.txt \
|
||||
--arg url "https://github.com/${{ github.repository }}/releases/tag/${{ env.RELEASE_VERSION }}" \
|
||||
'{
|
||||
text: (":rocket: New :runpod-new-whiteonpurple: Runpod worker-vllm release: *" + $version + "*"),
|
||||
blocks: [
|
||||
{
|
||||
"type": "section",
|
||||
"text": {
|
||||
"type": "mrkdwn",
|
||||
"text": ":banana-dance: *New Release — worker-vllm ${{ env.RELEASE_VERSION }}*\n*Docker:* `${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}`\n<https://github.com/${{ github.repository }}/releases/tag/${{ env.RELEASE_VERSION }}|View release on GitHub>"
|
||||
type: "section",
|
||||
text: {
|
||||
type: "mrkdwn",
|
||||
text: (":banana-dance: *New Release — worker-vllm " + $version + "*\n*Docker:* `" + $docker + "`\n<" + $url + "|View release on GitHub>")
|
||||
}
|
||||
},
|
||||
{
|
||||
type: "section",
|
||||
text: {
|
||||
type: "mrkdwn",
|
||||
text: ("*Release Notes:*\n" + $notes)
|
||||
}
|
||||
}
|
||||
]
|
||||
}'
|
||||
}' | curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d @-
|
||||
|
||||
@@ -6,6 +6,8 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
|
||||
|
||||
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
||||
|
||||
Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.19.1)
|
||||
|
||||
---
|
||||
|
||||
## Endpoint Configuration
|
||||
@@ -27,6 +29,7 @@ All behaviour is controlled through environment variables:
|
||||
| `REASONING_PARSER` | Parser for reasoning-capable models | | "deepseek_r1", "qwen3", "granite", "hunyuan_a13b" |
|
||||
| `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | Override served model name in API | | String |
|
||||
| `MAX_CONCURRENCY` | Maximum concurrent requests | 300 | Integer |
|
||||
| `ENFORCE_EAGER` | If True, we will disable CUDA graph and always execute the model in eager mode. If False, we will use CUDA graph and eager execution in hybrid for maximal performance and flexibility. | true | boolean (true or false) |
|
||||
|
||||
**Pass any vLLM engine arg** not listed above by setting an env var with the **UPPERCASED** field name (e.g. `MAX_MODEL_LEN=4096`, `ENABLE_CHUNKED_PREFILL=true`). The worker auto-discovers all `AsyncEngineArgs` fields from env. See the [vLLM engine args docs](https://docs.vllm.ai/en/latest/configuration/engine_args) for all available options.
|
||||
|
||||
|
||||
+11
-1
@@ -621,7 +621,7 @@
|
||||
"name": "Enforce Eager",
|
||||
"type": "boolean",
|
||||
"description": "Always use eager-mode PyTorch. If False (0), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility",
|
||||
"default": false,
|
||||
"default": true,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
@@ -795,6 +795,16 @@
|
||||
"default": "",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "PYTORCH_ALLOC_CONF",
|
||||
"input": {
|
||||
"name": "PyTorch Alloc Config",
|
||||
"type": "string",
|
||||
"description": "PyTorch allocation configuration, remove this if you want to use the default configuration",
|
||||
"default": "expandable_segments:True",
|
||||
"advanced": true
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
+1
-1
@@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/
|
||||
|
||||
# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels
|
||||
RUN uv pip install --system "packaging>=24.2" && \
|
||||
uv pip install --system "vllm[flashinfer]==0.17.1" --extra-index-url https://download.pytorch.org/whl/cu129
|
||||
uv pip install --system "vllm[flashinfer]==0.19.1" --extra-index-url https://download.pytorch.org/whl/cu129
|
||||
|
||||
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
||||
COPY builder/requirements.txt /requirements.txt
|
||||
|
||||
@@ -8,7 +8,8 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
|
||||
|
||||

|
||||
|
||||
Current vLLM version: [0.16.0](https://github.com/vllm-project/vllm/releases/tag/v0.16.0)
|
||||
Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.19.1)
|
||||
|
||||
|
||||
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
ray
|
||||
pandas
|
||||
pyarrow
|
||||
runpod
|
||||
runpod==1.9.0
|
||||
huggingface-hub
|
||||
lmcache==0.4.2
|
||||
packaging>=24.2
|
||||
@@ -9,7 +9,7 @@ typing-extensions>=4.8.0
|
||||
pydantic
|
||||
pydantic-settings
|
||||
hf-transfer
|
||||
transformers>=4.57.0,<5
|
||||
transformers>=5
|
||||
bitsandbytes>=0.45.0
|
||||
kernels
|
||||
torch-c-dlpack-ext
|
||||
|
||||
+64
-14
@@ -19,6 +19,7 @@ from vllm.entrypoints.openai.models.protocol import BaseModelPath, LoRAModulePat
|
||||
from vllm.entrypoints.openai.models.serving import OpenAIServingModels
|
||||
from vllm.entrypoints.openai.responses.protocol import ResponsesRequest, ResponsesResponse
|
||||
from vllm.entrypoints.openai.responses.serving import OpenAIServingResponses
|
||||
from vllm.entrypoints.serve.render.serving import OpenAIServingRender
|
||||
|
||||
from constants import DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MAX_CONCURRENCY, DEFAULT_MIN_BATCH_SIZE
|
||||
from engine_args import get_engine_args
|
||||
@@ -205,19 +206,48 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
self.raw_openai_output = bool(int(raw_output_env))
|
||||
|
||||
def _load_lora_adapters(self):
|
||||
adapters = []
|
||||
try:
|
||||
adapters = json.loads(os.getenv("LORA_MODULES", '[]'))
|
||||
except Exception as e:
|
||||
logging.info(f"---Initialized adapter json load error: {e}")
|
||||
lora_modules_env = os.getenv("LORA_MODULES", "")
|
||||
if not lora_modules_env:
|
||||
return []
|
||||
|
||||
for i, adapter in enumerate(adapters):
|
||||
try:
|
||||
adapters[i] = LoRAModulePath(**adapter)
|
||||
logging.info(f"---Initialized adapter: {adapter}")
|
||||
parsed = json.loads(lora_modules_env)
|
||||
except json.JSONDecodeError as e:
|
||||
logging.error(
|
||||
"LORA_MODULES could not be parsed as JSON: %s — no LoRA adapters loaded. Value: %r",
|
||||
e, lora_modules_env,
|
||||
)
|
||||
return []
|
||||
|
||||
# Accept a single adapter dict as well as an array
|
||||
if isinstance(parsed, dict):
|
||||
parsed = [parsed]
|
||||
|
||||
if not isinstance(parsed, list):
|
||||
logging.error(
|
||||
"LORA_MODULES must be a JSON array of adapter objects, got %s — no LoRA adapters loaded.",
|
||||
type(parsed).__name__,
|
||||
)
|
||||
return []
|
||||
|
||||
adapters = []
|
||||
for i, adapter in enumerate(parsed):
|
||||
try:
|
||||
adapters.append(LoRAModulePath(**adapter))
|
||||
logging.info("Loaded LoRA adapter config [%d]: %s", i, adapter)
|
||||
except Exception as e:
|
||||
logging.info(f"---Initialized adapter not worked: {e}")
|
||||
continue
|
||||
logging.error(
|
||||
"Failed to parse LoRA adapter at index %d: %s. Config: %r",
|
||||
i, e, adapter,
|
||||
)
|
||||
|
||||
if parsed and not adapters:
|
||||
logging.error(
|
||||
"LORA_MODULES specified %d adapter(s) but none could be loaded — "
|
||||
"OpenAI model name lookups for LoRA adapters will fail.",
|
||||
len(parsed),
|
||||
)
|
||||
|
||||
return adapters
|
||||
|
||||
async def _ensure_engines_initialized(self):
|
||||
@@ -252,10 +282,27 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
if self.tokenizer and hasattr(self.tokenizer, 'tokenizer'):
|
||||
chat_template = self.tokenizer.tokenizer.chat_template
|
||||
|
||||
self.openai_serving_render = OpenAIServingRender(
|
||||
model_config=self.llm.model_config,
|
||||
renderer=self.llm.renderer,
|
||||
io_processor=self.llm.io_processor,
|
||||
model_registry=self.serving_models.registry,
|
||||
request_logger=None,
|
||||
chat_template=chat_template,
|
||||
chat_template_content_format="auto",
|
||||
trust_request_chat_template=os.getenv('TRUST_REQUEST_CHAT_TEMPLATE', 'false').lower() == 'true',
|
||||
enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true',
|
||||
exclude_tools_when_tool_choice_none=os.getenv('EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE', 'false').lower() == 'true',
|
||||
tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None,
|
||||
reasoning_parser=os.getenv('REASONING_PARSER', "") or None,
|
||||
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
|
||||
)
|
||||
|
||||
self.chat_engine = OpenAIServingChat(
|
||||
engine_client=self.llm,
|
||||
models=self.serving_models,
|
||||
response_role=self.response_role,
|
||||
openai_serving_render=self.openai_serving_render,
|
||||
request_logger=None,
|
||||
chat_template=chat_template,
|
||||
chat_template_content_format="auto",
|
||||
@@ -268,20 +315,20 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
|
||||
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
||||
enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true',
|
||||
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
|
||||
)
|
||||
self.completion_engine = OpenAIServingCompletion(
|
||||
engine_client=self.llm,
|
||||
models=self.serving_models,
|
||||
openai_serving_render=self.openai_serving_render,
|
||||
request_logger=None,
|
||||
return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true',
|
||||
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
|
||||
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
||||
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
|
||||
)
|
||||
self.responses_engine = OpenAIServingResponses(
|
||||
engine_client=self.llm,
|
||||
models=self.serving_models,
|
||||
openai_serving_render=self.openai_serving_render,
|
||||
request_logger=None,
|
||||
chat_template=chat_template,
|
||||
chat_template_content_format="auto",
|
||||
@@ -293,12 +340,12 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
|
||||
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
|
||||
enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true',
|
||||
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
|
||||
)
|
||||
self.messages_engine = AnthropicServingMessages(
|
||||
engine_client=self.llm,
|
||||
models=self.serving_models,
|
||||
response_role=self.response_role,
|
||||
openai_serving_render=self.openai_serving_render,
|
||||
request_logger=None,
|
||||
chat_template=chat_template,
|
||||
chat_template_content_format="auto",
|
||||
@@ -311,7 +358,10 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
)
|
||||
|
||||
if hasattr(self.chat_engine, 'warmup'):
|
||||
await self.chat_engine.warmup()
|
||||
import asyncio
|
||||
result = self.chat_engine.warmup()
|
||||
if asyncio.iscoroutine(result):
|
||||
await result
|
||||
|
||||
async def generate(self, openai_request: JobInput):
|
||||
# Ensure engines are ready (no-op if already initialized at startup)
|
||||
|
||||
@@ -82,6 +82,16 @@ def _convert_env_value_to_field_type(value: str, field_name: str, field_type: ty
|
||||
if type(None) in (args or ()):
|
||||
return None
|
||||
raise ValueError("empty value not allowed for non-optional field")
|
||||
|
||||
# Union[bool, str, ...]: only coerce to bool for unambiguous literals;
|
||||
# otherwise preserve the string (e.g. hf_token="hf_abc..." must stay a str).
|
||||
if get_origin(field_type) is not None:
|
||||
union_types = [a for a in (get_args(field_type) or ()) if a is not type(None)]
|
||||
if bool in union_types and str in union_types:
|
||||
if str(val).lower() in ("true", "false", "1", "0", "yes", "no", "on", "off"):
|
||||
return str(val).lower() in ("true", "1", "yes", "on")
|
||||
return str(val)
|
||||
|
||||
effective_type = _resolve_field_type(field_type)
|
||||
# bool
|
||||
if effective_type is bool:
|
||||
@@ -342,6 +352,58 @@ def _sanitize_hf_overrides(hf_overrides: dict) -> dict | None:
|
||||
return result or None
|
||||
|
||||
|
||||
def _resolve_cached_model_path(model_name: str) -> str:
|
||||
"""Return a local snapshot path when the HF cache was stored with lowercase names.
|
||||
|
||||
Some model stores (e.g. RunPod pre-cached volumes) normalize repo IDs to
|
||||
lowercase. HuggingFace Hub stores caches as
|
||||
``models--{org}--{model}/snapshots/{hash}/`` preserving the original casing,
|
||||
so MODEL_NAME=Qwen/Qwen2.5-Coder-32B-Instruct-AWQ will miss a cache stored
|
||||
as ``models--qwen--qwen2.5-coder-32b-instruct-awq/``.
|
||||
|
||||
If the exact-case cache directory is absent but a lowercase variant exists,
|
||||
the latest snapshot path is returned so vLLM loads from disk rather than
|
||||
attempting a redundant download.
|
||||
"""
|
||||
if os.path.isabs(model_name):
|
||||
return model_name
|
||||
|
||||
cache_dir = (
|
||||
os.getenv("HUGGINGFACE_HUB_CACHE")
|
||||
or os.getenv("HF_HOME")
|
||||
or os.path.expanduser("~/.cache/huggingface/hub")
|
||||
)
|
||||
|
||||
folder_name = f"models--{model_name.replace('/', '--')}"
|
||||
|
||||
if os.path.isdir(os.path.join(cache_dir, folder_name)):
|
||||
return model_name
|
||||
|
||||
lower_dir = os.path.join(cache_dir, folder_name.lower())
|
||||
if not os.path.isdir(lower_dir):
|
||||
return model_name
|
||||
|
||||
snapshots_dir = os.path.join(lower_dir, "snapshots")
|
||||
if not os.path.isdir(snapshots_dir):
|
||||
return model_name
|
||||
|
||||
try:
|
||||
snapshots = sorted(os.listdir(snapshots_dir))
|
||||
except OSError:
|
||||
return model_name
|
||||
|
||||
if not snapshots:
|
||||
return model_name
|
||||
|
||||
resolved = os.path.join(snapshots_dir, snapshots[-1])
|
||||
logging.info(
|
||||
"MODEL_NAME %r not found at original casing in HF cache; "
|
||||
"resolved to lowercase cached snapshot at %r",
|
||||
model_name, resolved,
|
||||
)
|
||||
return resolved
|
||||
|
||||
|
||||
def get_local_args():
|
||||
"""
|
||||
Retrieve local arguments from a JSON file.
|
||||
@@ -517,4 +579,8 @@ def get_engine_args():
|
||||
if speculative_config:
|
||||
args["speculative_config"] = speculative_config
|
||||
|
||||
# Resolve lowercase HF cache paths (FDE-174)
|
||||
if args.get("model"):
|
||||
args["model"] = _resolve_cached_model_path(args["model"])
|
||||
|
||||
return AsyncEngineArgs(**args)
|
||||
|
||||
Reference in New Issue
Block a user