Merge branch 'main' into bug/281-hf-token

This commit is contained in:
velaraptor-runpod
2026-05-01 14:33:51 -05:00
9 changed files with 206 additions and 69 deletions
+28 -27
View File
@@ -9,59 +9,60 @@ on:
workflow_dispatch: workflow_dispatch:
permissions:
contents: write
pull-requests: write
jobs: jobs:
check_dep: check_dep:
runs-on: ubuntu-latest runs-on: ubuntu-latest
name: Check python requirements file and update name: Check python requirements file and update
steps: steps:
- name: Checkout - name: Checkout
uses: actions/checkout@v2 uses: actions/checkout@v4
- name: Check for new package version and update - name: Check for new package version and update
run: | run: |
echo "Fetching the current runpod version from requirements.txt..." echo "Fetching current runpod version from requirements.txt..."
# Get current version, allowing both == and ~= in the search pattern # Match runpod with any version specifier or no specifier at all
current_version=$(grep -oP 'runpod[~=]{1,2}\K[^"]+' ./builder/requirements.txt) current_version=$(grep -oP '^runpod([~>=!<]{1,2}\K[\d.]+)?' ./builder/requirements.txt | grep -oP '[\d.]+' || echo "")
echo "Current version: $current_version" echo "Current version: ${current_version:-unset}"
# Extract major and minor from current version echo "Fetching latest runpod version from PyPI..."
current_major_minor=$(echo $current_version | cut -d. -f1,2) new_version=$(curl -sf https://pypi.org/pypi/runpod/json | jq -r .info.version)
echo "Current major.minor: $current_major_minor"
echo "Fetching the latest runpod version from PyPI..."
# Get new version from PyPI
new_version=$(curl -s https://pypi.org/pypi/runpod/json | jq -r .info.version)
echo "NEW_VERSION_ENV=$new_version" >> $GITHUB_ENV echo "NEW_VERSION_ENV=$new_version" >> $GITHUB_ENV
echo "New version: $new_version" echo "New version: $new_version"
# Extract major and minor from new version
new_major_minor=$(echo $new_version | cut -d. -f1,2)
echo "New major.minor: $new_major_minor"
if [ -z "$new_version" ]; then if [ -z "$new_version" ]; then
echo "ERROR: Failed to fetch the new version from PyPI." echo "ERROR: Failed to fetch new version from PyPI."
exit 1 exit 1
fi fi
# Check if the major or minor version is different if [ -z "$current_version" ]; then
echo "No version pin found — pinning to $new_version."
else
current_major_minor=$(echo "$current_version" | cut -d. -f1,2)
new_major_minor=$(echo "$new_version" | cut -d. -f1,2)
echo "Current major.minor: $current_major_minor New major.minor: $new_major_minor"
if [ "$current_major_minor" = "$new_major_minor" ]; then if [ "$current_major_minor" = "$new_major_minor" ]; then
echo "No update needed. The new version ($new_major_minor) is within the allowed range (~= $current_major_minor)." echo "No update needed. New version ($new_version) is within ~= $current_major_minor range."
exit 0 exit 0
fi fi
echo "New major/minor detected ($new_major_minor). Updating requirements.txt..." echo "New major/minor detected ($new_major_minor). Updating requirements.txt..."
fi
# Update requirements.txt, preserving the existing constraint type (~= or ==) # Replace any `runpod`, `runpod==x`, `runpod~=x`, etc. with pinned version
sed -i "s/runpod[~=][^ ]*/runpod~=$new_version/" ./builder/requirements.txt sed -i "s|^runpod.*|runpod~=$new_version|" ./builder/requirements.txt
echo "requirements.txt has been updated." echo "requirements.txt updated."
- name: Create Pull Request - name: Create Pull Request
uses: peter-evans/create-pull-request@v3 uses: peter-evans/create-pull-request@v7
with: with:
token: ${{ secrets.GITHUB_TOKEN }} token: ${{ secrets.GITHUB_TOKEN }}
commit-message: Update runpod package version commit-message: "chore: update runpod to ${{ env.NEW_VERSION_ENV }}"
title: Update runpod package version title: "chore: update runpod to ${{ env.NEW_VERSION_ENV }}"
body: The package version has been updated to ${{ env.NEW_VERSION_ENV }} body: The `runpod` package has been updated to `${{ env.NEW_VERSION_ENV }}`.
branch: runpod-package-update branch: runpod-package-update
+36 -17
View File
@@ -3,7 +3,7 @@ name: Release
on: on:
push: push:
tags: tags:
- "v[0-9]+.[0-9]+.[0-9]+*" # Trigger on version tags like v1.0.0, v2.1.0, etc. - "v[0-9]+.[0-9]+.[0-9]+*"
workflow_dispatch: workflow_dispatch:
inputs: inputs:
version: version:
@@ -53,16 +53,13 @@ jobs:
# Determine version based on trigger type # Determine version based on trigger type
if [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then if [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then
# Manual trigger: use input version
VERSION="${{ github.event.inputs.version }}" VERSION="${{ github.event.inputs.version }}"
echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV elif [[ "${{ github.event_name }}" == "release" ]]; then
echo "IS_MANUAL_RELEASE=true" >> $GITHUB_ENV VERSION="${{ github.event.release.tag_name }}"
else else
# Tag trigger: use tag name (remove refs/tags/ prefix)
VERSION=${GITHUB_REF#refs/tags/} VERSION=${GITHUB_REF#refs/tags/}
echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV
echo "IS_MANUAL_RELEASE=false" >> $GITHUB_ENV
fi fi
echo "RELEASE_VERSION=${VERSION}" >> $GITHUB_ENV
- name: Build and push the images to Docker Hub - name: Build and push the images to Docker Hub
uses: docker/bake-action@v2 uses: docker/bake-action@v2
@@ -80,19 +77,41 @@ jobs:
echo "Version: ${{ env.RELEASE_VERSION }}" echo "Version: ${{ env.RELEASE_VERSION }}"
echo "Docker Image: ${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}" echo "Docker Image: ${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}"
- name: Fetch Release Notes
run: |
RESPONSE=$(curl -sf \
-H "Authorization: token ${{ github.token }}" \
"https://api.github.com/repos/${{ github.repository }}/releases/tags/${{ env.RELEASE_VERSION }}" 2>/dev/null) || true
if [[ -n "$RESPONSE" ]]; then
NOTES=$(echo "$RESPONSE" | jq -r '.body // empty')
fi
printf '%s' "${NOTES:-No release notes available.}" > /tmp/release_notes.txt
- name: Notify Slack - name: Notify Slack
run: | run: |
curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \ jq -n \
-H "Content-Type: application/json" \ --arg version "${{ env.RELEASE_VERSION }}" \
-d '{ --arg docker "${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}" \
"text": ":rocket: New :runpod-new-whiteonpurple: Runpod worker-vllm release: *${{ env.RELEASE_VERSION }}*", --rawfile notes /tmp/release_notes.txt \
"blocks": [ --arg url "https://github.com/${{ github.repository }}/releases/tag/${{ env.RELEASE_VERSION }}" \
'{
text: (":rocket: New :runpod-new-whiteonpurple: Runpod worker-vllm release: *" + $version + "*"),
blocks: [
{ {
"type": "section", type: "section",
"text": { text: {
"type": "mrkdwn", type: "mrkdwn",
"text": ":banana-dance: *New Release — worker-vllm ${{ env.RELEASE_VERSION }}*\n*Docker:* `${{ env.DOCKERHUB_REPO }}/${{ env.DOCKERHUB_IMG }}:${{ env.RELEASE_VERSION }}`\n<https://github.com/${{ github.repository }}/releases/tag/${{ env.RELEASE_VERSION }}|View release on GitHub>" text: (":banana-dance: *New Release — worker-vllm " + $version + "*\n*Docker:* `" + $docker + "`\n<" + $url + "|View release on GitHub>")
}
},
{
type: "section",
text: {
type: "mrkdwn",
text: ("*Release Notes:*\n" + $notes)
} }
} }
] ]
}' }' | curl -sf -X POST "${{ secrets.SLACK_WEBHOOK_URL }}" \
-H "Content-Type: application/json" \
-d @-
+3
View File
@@ -6,6 +6,8 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
[![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) [![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.19.1)
--- ---
## Endpoint Configuration ## Endpoint Configuration
@@ -27,6 +29,7 @@ All behaviour is controlled through environment variables:
| `REASONING_PARSER` | Parser for reasoning-capable models | | "deepseek_r1", "qwen3", "granite", "hunyuan_a13b" | | `REASONING_PARSER` | Parser for reasoning-capable models | | "deepseek_r1", "qwen3", "granite", "hunyuan_a13b" |
| `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | Override served model name in API | | String | | `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | Override served model name in API | | String |
| `MAX_CONCURRENCY` | Maximum concurrent requests | 300 | Integer | | `MAX_CONCURRENCY` | Maximum concurrent requests | 300 | Integer |
| `ENFORCE_EAGER` | If True, we will disable CUDA graph and always execute the model in eager mode. If False, we will use CUDA graph and eager execution in hybrid for maximal performance and flexibility. | true | boolean (true or false) |
**Pass any vLLM engine arg** not listed above by setting an env var with the **UPPERCASED** field name (e.g. `MAX_MODEL_LEN=4096`, `ENABLE_CHUNKED_PREFILL=true`). The worker auto-discovers all `AsyncEngineArgs` fields from env. See the [vLLM engine args docs](https://docs.vllm.ai/en/latest/configuration/engine_args) for all available options. **Pass any vLLM engine arg** not listed above by setting an env var with the **UPPERCASED** field name (e.g. `MAX_MODEL_LEN=4096`, `ENABLE_CHUNKED_PREFILL=true`). The worker auto-discovers all `AsyncEngineArgs` fields from env. See the [vLLM engine args docs](https://docs.vllm.ai/en/latest/configuration/engine_args) for all available options.
+11 -1
View File
@@ -621,7 +621,7 @@
"name": "Enforce Eager", "name": "Enforce Eager",
"type": "boolean", "type": "boolean",
"description": "Always use eager-mode PyTorch. If False (0), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility", "description": "Always use eager-mode PyTorch. If False (0), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility",
"default": false, "default": true,
"advanced": true "advanced": true
} }
}, },
@@ -795,6 +795,16 @@
"default": "", "default": "",
"advanced": true "advanced": true
} }
},
{
"key": "PYTORCH_ALLOC_CONF",
"input": {
"name": "PyTorch Alloc Config",
"type": "string",
"description": "PyTorch allocation configuration, remove this if you want to use the default configuration",
"default": "expandable_segments:True",
"advanced": true
}
} }
] ]
} }
+1 -1
View File
@@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/
# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels # Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels
RUN uv pip install --system "packaging>=24.2" && \ RUN uv pip install --system "packaging>=24.2" && \
uv pip install --system "vllm[flashinfer]==0.17.1" --extra-index-url https://download.pytorch.org/whl/cu129 uv pip install --system "vllm[flashinfer]==0.19.1" --extra-index-url https://download.pytorch.org/whl/cu129
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
COPY builder/requirements.txt /requirements.txt COPY builder/requirements.txt /requirements.txt
+2 -1
View File
@@ -8,7 +8,8 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png) ![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png)
Current vLLM version: [0.16.0](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) Current vLLM version: [0.19.1](https://github.com/vllm-project/vllm/releases/tag/v0.19.1)
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep) > Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
+2 -2
View File
@@ -1,7 +1,7 @@
ray ray
pandas pandas
pyarrow pyarrow
runpod runpod==1.9.0
huggingface-hub huggingface-hub
lmcache==0.4.2 lmcache==0.4.2
packaging>=24.2 packaging>=24.2
@@ -9,7 +9,7 @@ typing-extensions>=4.8.0
pydantic pydantic
pydantic-settings pydantic-settings
hf-transfer hf-transfer
transformers>=4.57.0,<5 transformers>=5
bitsandbytes>=0.45.0 bitsandbytes>=0.45.0
kernels kernels
torch-c-dlpack-ext torch-c-dlpack-ext
+60 -13
View File
@@ -19,6 +19,7 @@ from vllm.entrypoints.openai.models.protocol import BaseModelPath, LoRAModulePat
from vllm.entrypoints.openai.models.serving import OpenAIServingModels from vllm.entrypoints.openai.models.serving import OpenAIServingModels
from vllm.entrypoints.openai.responses.protocol import ResponsesRequest, ResponsesResponse from vllm.entrypoints.openai.responses.protocol import ResponsesRequest, ResponsesResponse
from vllm.entrypoints.openai.responses.serving import OpenAIServingResponses from vllm.entrypoints.openai.responses.serving import OpenAIServingResponses
from vllm.entrypoints.serve.render.serving import OpenAIServingRender
from constants import DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MAX_CONCURRENCY, DEFAULT_MIN_BATCH_SIZE from constants import DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MAX_CONCURRENCY, DEFAULT_MIN_BATCH_SIZE
from engine_args import get_engine_args from engine_args import get_engine_args
@@ -205,19 +206,48 @@ class OpenAIvLLMEngine(vLLMEngine):
self.raw_openai_output = bool(int(raw_output_env)) self.raw_openai_output = bool(int(raw_output_env))
def _load_lora_adapters(self): def _load_lora_adapters(self):
adapters = [] lora_modules_env = os.getenv("LORA_MODULES", "")
try: if not lora_modules_env:
adapters = json.loads(os.getenv("LORA_MODULES", '[]')) return []
except Exception as e:
logging.info(f"---Initialized adapter json load error: {e}")
for i, adapter in enumerate(adapters):
try: try:
adapters[i] = LoRAModulePath(**adapter) parsed = json.loads(lora_modules_env)
logging.info(f"---Initialized adapter: {adapter}") except json.JSONDecodeError as e:
logging.error(
"LORA_MODULES could not be parsed as JSON: %s — no LoRA adapters loaded. Value: %r",
e, lora_modules_env,
)
return []
# Accept a single adapter dict as well as an array
if isinstance(parsed, dict):
parsed = [parsed]
if not isinstance(parsed, list):
logging.error(
"LORA_MODULES must be a JSON array of adapter objects, got %s — no LoRA adapters loaded.",
type(parsed).__name__,
)
return []
adapters = []
for i, adapter in enumerate(parsed):
try:
adapters.append(LoRAModulePath(**adapter))
logging.info("Loaded LoRA adapter config [%d]: %s", i, adapter)
except Exception as e: except Exception as e:
logging.info(f"---Initialized adapter not worked: {e}") logging.error(
continue "Failed to parse LoRA adapter at index %d: %s. Config: %r",
i, e, adapter,
)
if parsed and not adapters:
logging.error(
"LORA_MODULES specified %d adapter(s) but none could be loaded — "
"OpenAI model name lookups for LoRA adapters will fail.",
len(parsed),
)
return adapters return adapters
async def _ensure_engines_initialized(self): async def _ensure_engines_initialized(self):
@@ -252,10 +282,27 @@ class OpenAIvLLMEngine(vLLMEngine):
if self.tokenizer and hasattr(self.tokenizer, 'tokenizer'): if self.tokenizer and hasattr(self.tokenizer, 'tokenizer'):
chat_template = self.tokenizer.tokenizer.chat_template chat_template = self.tokenizer.tokenizer.chat_template
self.openai_serving_render = OpenAIServingRender(
model_config=self.llm.model_config,
renderer=self.llm.renderer,
io_processor=self.llm.io_processor,
model_registry=self.serving_models.registry,
request_logger=None,
chat_template=chat_template,
chat_template_content_format="auto",
trust_request_chat_template=os.getenv('TRUST_REQUEST_CHAT_TEMPLATE', 'false').lower() == 'true',
enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true',
exclude_tools_when_tool_choice_none=os.getenv('EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE', 'false').lower() == 'true',
tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None,
reasoning_parser=os.getenv('REASONING_PARSER', "") or None,
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
)
self.chat_engine = OpenAIServingChat( self.chat_engine = OpenAIServingChat(
engine_client=self.llm, engine_client=self.llm,
models=self.serving_models, models=self.serving_models,
response_role=self.response_role, response_role=self.response_role,
openai_serving_render=self.openai_serving_render,
request_logger=None, request_logger=None,
chat_template=chat_template, chat_template=chat_template,
chat_template_content_format="auto", chat_template_content_format="auto",
@@ -268,20 +315,20 @@ class OpenAIvLLMEngine(vLLMEngine):
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true', enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true', enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true', enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true',
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
) )
self.completion_engine = OpenAIServingCompletion( self.completion_engine = OpenAIServingCompletion(
engine_client=self.llm, engine_client=self.llm,
models=self.serving_models, models=self.serving_models,
openai_serving_render=self.openai_serving_render,
request_logger=None, request_logger=None,
return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true', return_tokens_as_token_ids=os.getenv('RETURN_TOKENS_AS_TOKEN_IDS', 'false').lower() == 'true',
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true', enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true', enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
) )
self.responses_engine = OpenAIServingResponses( self.responses_engine = OpenAIServingResponses(
engine_client=self.llm, engine_client=self.llm,
models=self.serving_models, models=self.serving_models,
openai_serving_render=self.openai_serving_render,
request_logger=None, request_logger=None,
chat_template=chat_template, chat_template=chat_template,
chat_template_content_format="auto", chat_template_content_format="auto",
@@ -293,12 +340,12 @@ class OpenAIvLLMEngine(vLLMEngine):
enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true', enable_prompt_tokens_details=os.getenv('ENABLE_PROMPT_TOKENS_DETAILS', 'false').lower() == 'true',
enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true', enable_force_include_usage=os.getenv('ENABLE_FORCE_INCLUDE_USAGE', 'false').lower() == 'true',
enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true', enable_log_outputs=os.getenv('ENABLE_LOG_OUTPUTS', 'false').lower() == 'true',
log_error_stack=os.getenv('LOG_ERROR_STACK', 'false').lower() == 'true',
) )
self.messages_engine = AnthropicServingMessages( self.messages_engine = AnthropicServingMessages(
engine_client=self.llm, engine_client=self.llm,
models=self.serving_models, models=self.serving_models,
response_role=self.response_role, response_role=self.response_role,
openai_serving_render=self.openai_serving_render,
request_logger=None, request_logger=None,
chat_template=chat_template, chat_template=chat_template,
chat_template_content_format="auto", chat_template_content_format="auto",
+56
View File
@@ -352,6 +352,58 @@ def _sanitize_hf_overrides(hf_overrides: dict) -> dict | None:
return result or None return result or None
def _resolve_cached_model_path(model_name: str) -> str:
"""Return a local snapshot path when the HF cache was stored with lowercase names.
Some model stores (e.g. RunPod pre-cached volumes) normalize repo IDs to
lowercase. HuggingFace Hub stores caches as
``models--{org}--{model}/snapshots/{hash}/`` preserving the original casing,
so MODEL_NAME=Qwen/Qwen2.5-Coder-32B-Instruct-AWQ will miss a cache stored
as ``models--qwen--qwen2.5-coder-32b-instruct-awq/``.
If the exact-case cache directory is absent but a lowercase variant exists,
the latest snapshot path is returned so vLLM loads from disk rather than
attempting a redundant download.
"""
if os.path.isabs(model_name):
return model_name
cache_dir = (
os.getenv("HUGGINGFACE_HUB_CACHE")
or os.getenv("HF_HOME")
or os.path.expanduser("~/.cache/huggingface/hub")
)
folder_name = f"models--{model_name.replace('/', '--')}"
if os.path.isdir(os.path.join(cache_dir, folder_name)):
return model_name
lower_dir = os.path.join(cache_dir, folder_name.lower())
if not os.path.isdir(lower_dir):
return model_name
snapshots_dir = os.path.join(lower_dir, "snapshots")
if not os.path.isdir(snapshots_dir):
return model_name
try:
snapshots = sorted(os.listdir(snapshots_dir))
except OSError:
return model_name
if not snapshots:
return model_name
resolved = os.path.join(snapshots_dir, snapshots[-1])
logging.info(
"MODEL_NAME %r not found at original casing in HF cache; "
"resolved to lowercase cached snapshot at %r",
model_name, resolved,
)
return resolved
def get_local_args(): def get_local_args():
""" """
Retrieve local arguments from a JSON file. Retrieve local arguments from a JSON file.
@@ -527,4 +579,8 @@ def get_engine_args():
if speculative_config: if speculative_config:
args["speculative_config"] = speculative_config args["speculative_config"] = speculative_config
# Resolve lowercase HF cache paths (FDE-174)
if args.get("model"):
args["model"] = _resolve_cached_model_path(args["model"])
return AsyncEngineArgs(**args) return AsyncEngineArgs(**args)