fix: serve original model name when hf cache dir is lowercased (#310)

the fde-174 cache resolver rewrites engine_args.model to an on-disk
snapshot path when the model is found only under a lowercased hf cache
dir. the openai served model name is derived from engine_args.model, so
it silently became the filesystem path and requests using the real repo
id returned 404.

set served_model_name to the original repo id whenever the model is
rewritten to a path, unless an explicit served name (or
OPENAI_SERVED_MODEL_NAME_OVERRIDE) is provided.

add the first python tests in the repo (tests/) covering the cache-path
resolution and served-name decoupling, plus a Tests github workflow that
runs pytest on prs and pushes to main. vllm/torch are stubbed when absent
so the suite runs on a plain cpu runner.
This commit is contained in:
Tim Pietrusky
2026-06-19 19:04:45 +02:00
parent 1b3228a2dc
commit fcdc799e0d
5 changed files with 248 additions and 1 deletions
+32
View File
@@ -0,0 +1,32 @@
name: Tests
on:
pull_request:
branches:
- "**"
push:
branches:
- "main"
permissions:
contents: read
jobs:
pytest:
runs-on: ubuntu-latest
steps:
- name: Checkout
uses: actions/checkout@v4
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: "3.11"
- name: Install test dependencies
run: |
python -m pip install --upgrade pip
pip install -r tests/requirements.txt
- name: Run unit tests
run: python -m pytest tests -v
+8 -1
View File
@@ -601,6 +601,13 @@ def get_engine_args():
# Resolve lowercase HF cache paths (FDE-174) # Resolve lowercase HF cache paths (FDE-174)
if args.get("model"): if args.get("model"):
args["model"] = _resolve_cached_model_path(args["model"]) original_model = args["model"]
args["model"] = _resolve_cached_model_path(original_model)
# When the model was rewritten to an on-disk snapshot path, keep serving
# under the original repo id so the OpenAI API model name does not become
# a filesystem path (issue #310). An explicit served_model_name (or the
# OPENAI_SERVED_MODEL_NAME_OVERRIDE handled downstream) still wins.
if args["model"] != original_model and not args.get("served_model_name"):
args["served_model_name"] = original_model
return AsyncEngineArgs(**args) return AsyncEngineArgs(**args)
+81
View File
@@ -0,0 +1,81 @@
"""Shared test fixtures.
``src/engine_args.py`` hard-imports ``vllm`` (and a tensorizer submodule) and
``torch.cuda``. Both are only installed inside the GPU Docker image, so when the
tests run on a machine without them we install lightweight stubs. When the real
packages *are* available (e.g. CI inside the worker image) the stubs are skipped
and the real ones are used instead.
"""
import sys
import types
from dataclasses import dataclass
from typing import Optional, Union, List
def _install_torch_stub():
try:
import torch # noqa: F401
return # real torch present, nothing to stub
except Exception:
pass
torch = types.ModuleType("torch")
cuda = types.ModuleType("torch.cuda")
# No GPU in the test environment -> 0 devices (skips tensor-parallel setup).
cuda.device_count = lambda: 0
torch.cuda = cuda
sys.modules["torch"] = torch
sys.modules["torch.cuda"] = cuda
def _install_vllm_stub():
try:
import vllm # noqa: F401
return # real vLLM present, nothing to stub
except Exception:
pass
vllm = types.ModuleType("vllm")
@dataclass
class AsyncEngineArgs:
# Only the fields the worker actually sets/reads need to exist here;
# get_engine_args() filters args down to AsyncEngineArgs.__dataclass_fields__
# before construction, so unknown keys are dropped rather than passed.
model: Optional[str] = None
served_model_name: Optional[Union[str, List[str]]] = None
revision: Optional[str] = None
tokenizer: Optional[str] = None
trust_remote_code: bool = False
max_model_len: Optional[int] = None
max_num_batched_tokens: Optional[int] = None
disable_log_stats: bool = False
gpu_memory_utilization: float = 0.9
tensor_parallel_size: int = 1
max_parallel_loading_workers: Optional[int] = None
kv_cache_dtype: Optional[str] = None
vllm.AsyncEngineArgs = AsyncEngineArgs
sys.modules["vllm"] = vllm
# vllm.model_executor.model_loader.tensorizer.TensorizerConfig
model_executor = types.ModuleType("vllm.model_executor")
model_loader = types.ModuleType("vllm.model_executor.model_loader")
tensorizer = types.ModuleType("vllm.model_executor.model_loader.tensorizer")
class TensorizerConfig: # pragma: no cover - placeholder
def __init__(self, *args, **kwargs):
pass
tensorizer.TensorizerConfig = TensorizerConfig
model_loader.tensorizer = tensorizer
model_executor.model_loader = model_loader
vllm.model_executor = model_executor
sys.modules["vllm.model_executor"] = model_executor
sys.modules["vllm.model_executor.model_loader"] = model_loader
sys.modules["vllm.model_executor.model_loader.tensorizer"] = tensorizer
_install_torch_stub()
_install_vllm_stub()
+3
View File
@@ -0,0 +1,3 @@
# Test-only dependencies. vllm/torch are stubbed in conftest.py when absent,
# so the unit tests run on a plain CPU runner without the GPU image.
pytest>=8,<10
+124
View File
@@ -0,0 +1,124 @@
"""Tests for HF cache path resolution and served-model-name decoupling.
Regression coverage for issue #310: when MODEL_NAME is served from a lowercased
HF cache dir, the cache resolver rewrites engine_args.model to a snapshot path.
The served model name must stay the original repo id, not the path.
"""
import os
import pytest
from src import engine_args
from src.engine_args import _resolve_cached_model_path, get_engine_args
MODEL = "Qwen/Qwen3.6-27B-FP8"
SNAPSHOT_HASH = "e89b16ebf1988b3d6befa7de50abc2d76f26eb09"
def _make_cache(root, folder_name, snapshot=SNAPSHOT_HASH):
"""Create a HF-style ``models--…/snapshots/<hash>/`` dir and return its path."""
snap_dir = os.path.join(root, folder_name, "snapshots", snapshot)
os.makedirs(snap_dir)
return snap_dir
def _is_case_sensitive_fs(path):
"""The lowercase-cache resolution only matters on case-sensitive filesystems.
On macOS (APFS, case-insensitive by default) ``models--Qwen--…`` and
``models--qwen--…`` collide, so the resolver always sees the exact-case dir
as present. Production runs on Linux (case-sensitive), which is what these
tests exercise.
"""
probe = os.path.join(path, "CaseProbe")
open(probe, "w").close()
try:
return not os.path.exists(os.path.join(path, "caseprobe"))
finally:
os.remove(probe)
requires_case_sensitive_fs = pytest.mark.skipif(
not _is_case_sensitive_fs(os.environ.get("TMPDIR", "/tmp")),
reason="lowercase HF cache resolution only applies on case-sensitive filesystems",
)
@pytest.fixture
def hf_cache(tmp_path, monkeypatch):
cache = tmp_path / "hub"
cache.mkdir()
monkeypatch.setenv("HUGGINGFACE_HUB_CACHE", str(cache))
# Make sure HF_HOME does not shadow the explicit cache dir during the test.
monkeypatch.delenv("HF_HOME", raising=False)
return cache
class TestResolveCachedModelPath:
def test_exact_case_dir_returns_repo_id(self, hf_cache):
_make_cache(str(hf_cache), "models--Qwen--Qwen3.6-27B-FP8")
assert _resolve_cached_model_path(MODEL) == MODEL
def test_no_cache_returns_repo_id(self, hf_cache):
assert _resolve_cached_model_path(MODEL) == MODEL
def test_absolute_path_passthrough(self, hf_cache):
path = "/runpod-volume/some/local/model"
assert _resolve_cached_model_path(path) == path
@requires_case_sensitive_fs
def test_lowercase_dir_returns_snapshot_path(self, hf_cache):
snap = _make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")
assert _resolve_cached_model_path(MODEL) == snap
def test_lowercase_dir_without_snapshots_returns_repo_id(self, hf_cache):
# Dir exists but has no snapshots subdir -> nothing to resolve to.
os.makedirs(os.path.join(str(hf_cache), "models--qwen--qwen3.6-27b-fp8"))
assert _resolve_cached_model_path(MODEL) == MODEL
@requires_case_sensitive_fs
def test_lowercase_dir_picks_latest_snapshot(self, hf_cache):
folder = "models--qwen--qwen3.6-27b-fp8"
_make_cache(str(hf_cache), folder, snapshot="aaaa")
latest = _make_cache(str(hf_cache), folder, snapshot="zzzz")
assert _resolve_cached_model_path(MODEL) == latest
class TestGetEngineArgsServedName:
"""Issue #310: served name must be decoupled from the resolved on-disk path."""
@pytest.fixture(autouse=True)
def base_env(self, monkeypatch):
# Avoid the network branch in _resolve_max_model_len.
monkeypatch.setenv("MAX_NUM_BATCHED_TOKENS", "2048")
monkeypatch.delenv("SERVED_MODEL_NAME", raising=False)
@requires_case_sensitive_fs
def test_served_name_is_repo_id_when_path_rewritten(self, hf_cache, monkeypatch):
snap = _make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")
monkeypatch.setenv("MODEL_NAME", MODEL)
result = get_engine_args()
assert result.model == snap # weights load from the lowercase cache
assert result.served_model_name == MODEL # API still serves the repo id
def test_served_name_untouched_when_no_rewrite(self, hf_cache, monkeypatch):
_make_cache(str(hf_cache), "models--Qwen--Qwen3.6-27B-FP8")
monkeypatch.setenv("MODEL_NAME", MODEL)
result = get_engine_args()
assert result.model == MODEL
assert result.served_model_name is None
def test_explicit_served_name_not_overridden(self, hf_cache, monkeypatch):
_make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")
monkeypatch.setenv("MODEL_NAME", MODEL)
monkeypatch.setenv("SERVED_MODEL_NAME", "custom-name")
result = get_engine_args()
assert result.served_model_name == "custom-name"