From fcdc799e0dd480c49f37e66871dfe22f10cc1a9e Mon Sep 17 00:00:00 2001 From: Tim Pietrusky Date: Fri, 19 Jun 2026 18:59:05 +0200 Subject: [PATCH 1/3] fix: serve original model name when hf cache dir is lowercased (#310) the fde-174 cache resolver rewrites engine_args.model to an on-disk snapshot path when the model is found only under a lowercased hf cache dir. the openai served model name is derived from engine_args.model, so it silently became the filesystem path and requests using the real repo id returned 404. set served_model_name to the original repo id whenever the model is rewritten to a path, unless an explicit served name (or OPENAI_SERVED_MODEL_NAME_OVERRIDE) is provided. add the first python tests in the repo (tests/) covering the cache-path resolution and served-name decoupling, plus a Tests github workflow that runs pytest on prs and pushes to main. vllm/torch are stubbed when absent so the suite runs on a plain cpu runner. --- .github/workflows/test.yml | 32 ++++++++++ src/engine_args.py | 9 ++- tests/conftest.py | 81 ++++++++++++++++++++++++ tests/requirements.txt | 3 + tests/test_engine_args.py | 124 +++++++++++++++++++++++++++++++++++++ 5 files changed, 248 insertions(+), 1 deletion(-) create mode 100644 .github/workflows/test.yml create mode 100644 tests/conftest.py create mode 100644 tests/requirements.txt create mode 100644 tests/test_engine_args.py diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml new file mode 100644 index 0000000..66aba32 --- /dev/null +++ b/.github/workflows/test.yml @@ -0,0 +1,32 @@ +name: Tests + +on: + pull_request: + branches: + - "**" + push: + branches: + - "main" + +permissions: + contents: read + +jobs: + pytest: + runs-on: ubuntu-latest + steps: + - name: Checkout + uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: "3.11" + + - name: Install test dependencies + run: | + python -m pip install --upgrade pip + pip install -r tests/requirements.txt + + - name: Run unit tests + run: python -m pytest tests -v diff --git a/src/engine_args.py b/src/engine_args.py index 3df33df..5749367 100644 --- a/src/engine_args.py +++ b/src/engine_args.py @@ -601,6 +601,13 @@ def get_engine_args(): # Resolve lowercase HF cache paths (FDE-174) if args.get("model"): - args["model"] = _resolve_cached_model_path(args["model"]) + original_model = args["model"] + args["model"] = _resolve_cached_model_path(original_model) + # When the model was rewritten to an on-disk snapshot path, keep serving + # under the original repo id so the OpenAI API model name does not become + # a filesystem path (issue #310). An explicit served_model_name (or the + # OPENAI_SERVED_MODEL_NAME_OVERRIDE handled downstream) still wins. + if args["model"] != original_model and not args.get("served_model_name"): + args["served_model_name"] = original_model return AsyncEngineArgs(**args) diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 0000000..d0d56f4 --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,81 @@ +"""Shared test fixtures. + +``src/engine_args.py`` hard-imports ``vllm`` (and a tensorizer submodule) and +``torch.cuda``. Both are only installed inside the GPU Docker image, so when the +tests run on a machine without them we install lightweight stubs. When the real +packages *are* available (e.g. CI inside the worker image) the stubs are skipped +and the real ones are used instead. +""" + +import sys +import types +from dataclasses import dataclass +from typing import Optional, Union, List + + +def _install_torch_stub(): + try: + import torch # noqa: F401 + return # real torch present, nothing to stub + except Exception: + pass + + torch = types.ModuleType("torch") + cuda = types.ModuleType("torch.cuda") + # No GPU in the test environment -> 0 devices (skips tensor-parallel setup). + cuda.device_count = lambda: 0 + torch.cuda = cuda + sys.modules["torch"] = torch + sys.modules["torch.cuda"] = cuda + + +def _install_vllm_stub(): + try: + import vllm # noqa: F401 + return # real vLLM present, nothing to stub + except Exception: + pass + + vllm = types.ModuleType("vllm") + + @dataclass + class AsyncEngineArgs: + # Only the fields the worker actually sets/reads need to exist here; + # get_engine_args() filters args down to AsyncEngineArgs.__dataclass_fields__ + # before construction, so unknown keys are dropped rather than passed. + model: Optional[str] = None + served_model_name: Optional[Union[str, List[str]]] = None + revision: Optional[str] = None + tokenizer: Optional[str] = None + trust_remote_code: bool = False + max_model_len: Optional[int] = None + max_num_batched_tokens: Optional[int] = None + disable_log_stats: bool = False + gpu_memory_utilization: float = 0.9 + tensor_parallel_size: int = 1 + max_parallel_loading_workers: Optional[int] = None + kv_cache_dtype: Optional[str] = None + + vllm.AsyncEngineArgs = AsyncEngineArgs + sys.modules["vllm"] = vllm + + # vllm.model_executor.model_loader.tensorizer.TensorizerConfig + model_executor = types.ModuleType("vllm.model_executor") + model_loader = types.ModuleType("vllm.model_executor.model_loader") + tensorizer = types.ModuleType("vllm.model_executor.model_loader.tensorizer") + + class TensorizerConfig: # pragma: no cover - placeholder + def __init__(self, *args, **kwargs): + pass + + tensorizer.TensorizerConfig = TensorizerConfig + model_loader.tensorizer = tensorizer + model_executor.model_loader = model_loader + vllm.model_executor = model_executor + sys.modules["vllm.model_executor"] = model_executor + sys.modules["vllm.model_executor.model_loader"] = model_loader + sys.modules["vllm.model_executor.model_loader.tensorizer"] = tensorizer + + +_install_torch_stub() +_install_vllm_stub() diff --git a/tests/requirements.txt b/tests/requirements.txt new file mode 100644 index 0000000..ce1cb0d --- /dev/null +++ b/tests/requirements.txt @@ -0,0 +1,3 @@ +# Test-only dependencies. vllm/torch are stubbed in conftest.py when absent, +# so the unit tests run on a plain CPU runner without the GPU image. +pytest>=8,<10 diff --git a/tests/test_engine_args.py b/tests/test_engine_args.py new file mode 100644 index 0000000..dce4d24 --- /dev/null +++ b/tests/test_engine_args.py @@ -0,0 +1,124 @@ +"""Tests for HF cache path resolution and served-model-name decoupling. + +Regression coverage for issue #310: when MODEL_NAME is served from a lowercased +HF cache dir, the cache resolver rewrites engine_args.model to a snapshot path. +The served model name must stay the original repo id, not the path. +""" + +import os + +import pytest + +from src import engine_args +from src.engine_args import _resolve_cached_model_path, get_engine_args + + +MODEL = "Qwen/Qwen3.6-27B-FP8" +SNAPSHOT_HASH = "e89b16ebf1988b3d6befa7de50abc2d76f26eb09" + + +def _make_cache(root, folder_name, snapshot=SNAPSHOT_HASH): + """Create a HF-style ``models--…/snapshots//`` dir and return its path.""" + snap_dir = os.path.join(root, folder_name, "snapshots", snapshot) + os.makedirs(snap_dir) + return snap_dir + + +def _is_case_sensitive_fs(path): + """The lowercase-cache resolution only matters on case-sensitive filesystems. + + On macOS (APFS, case-insensitive by default) ``models--Qwen--…`` and + ``models--qwen--…`` collide, so the resolver always sees the exact-case dir + as present. Production runs on Linux (case-sensitive), which is what these + tests exercise. + """ + probe = os.path.join(path, "CaseProbe") + open(probe, "w").close() + try: + return not os.path.exists(os.path.join(path, "caseprobe")) + finally: + os.remove(probe) + + +requires_case_sensitive_fs = pytest.mark.skipif( + not _is_case_sensitive_fs(os.environ.get("TMPDIR", "/tmp")), + reason="lowercase HF cache resolution only applies on case-sensitive filesystems", +) + + +@pytest.fixture +def hf_cache(tmp_path, monkeypatch): + cache = tmp_path / "hub" + cache.mkdir() + monkeypatch.setenv("HUGGINGFACE_HUB_CACHE", str(cache)) + # Make sure HF_HOME does not shadow the explicit cache dir during the test. + monkeypatch.delenv("HF_HOME", raising=False) + return cache + + +class TestResolveCachedModelPath: + def test_exact_case_dir_returns_repo_id(self, hf_cache): + _make_cache(str(hf_cache), "models--Qwen--Qwen3.6-27B-FP8") + assert _resolve_cached_model_path(MODEL) == MODEL + + def test_no_cache_returns_repo_id(self, hf_cache): + assert _resolve_cached_model_path(MODEL) == MODEL + + def test_absolute_path_passthrough(self, hf_cache): + path = "/runpod-volume/some/local/model" + assert _resolve_cached_model_path(path) == path + + @requires_case_sensitive_fs + def test_lowercase_dir_returns_snapshot_path(self, hf_cache): + snap = _make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8") + assert _resolve_cached_model_path(MODEL) == snap + + def test_lowercase_dir_without_snapshots_returns_repo_id(self, hf_cache): + # Dir exists but has no snapshots subdir -> nothing to resolve to. + os.makedirs(os.path.join(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")) + assert _resolve_cached_model_path(MODEL) == MODEL + + @requires_case_sensitive_fs + def test_lowercase_dir_picks_latest_snapshot(self, hf_cache): + folder = "models--qwen--qwen3.6-27b-fp8" + _make_cache(str(hf_cache), folder, snapshot="aaaa") + latest = _make_cache(str(hf_cache), folder, snapshot="zzzz") + assert _resolve_cached_model_path(MODEL) == latest + + +class TestGetEngineArgsServedName: + """Issue #310: served name must be decoupled from the resolved on-disk path.""" + + @pytest.fixture(autouse=True) + def base_env(self, monkeypatch): + # Avoid the network branch in _resolve_max_model_len. + monkeypatch.setenv("MAX_NUM_BATCHED_TOKENS", "2048") + monkeypatch.delenv("SERVED_MODEL_NAME", raising=False) + + @requires_case_sensitive_fs + def test_served_name_is_repo_id_when_path_rewritten(self, hf_cache, monkeypatch): + snap = _make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8") + monkeypatch.setenv("MODEL_NAME", MODEL) + + result = get_engine_args() + + assert result.model == snap # weights load from the lowercase cache + assert result.served_model_name == MODEL # API still serves the repo id + + def test_served_name_untouched_when_no_rewrite(self, hf_cache, monkeypatch): + _make_cache(str(hf_cache), "models--Qwen--Qwen3.6-27B-FP8") + monkeypatch.setenv("MODEL_NAME", MODEL) + + result = get_engine_args() + + assert result.model == MODEL + assert result.served_model_name is None + + def test_explicit_served_name_not_overridden(self, hf_cache, monkeypatch): + _make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8") + monkeypatch.setenv("MODEL_NAME", MODEL) + monkeypatch.setenv("SERVED_MODEL_NAME", "custom-name") + + result = get_engine_args() + + assert result.served_model_name == "custom-name" From b11c91722cf0e2aa600eccddc84985bb6b25028c Mon Sep 17 00:00:00 2001 From: Tim Pietrusky Date: Fri, 19 Jun 2026 19:02:49 +0200 Subject: [PATCH 2/3] test: complete vllm stub so src.utils imports under py<3.14 src.utils uses ErrorResponse (a vllm import) as a module-level return annotation, evaluated eagerly on python <3.14. the vllm stub lacked it, so collection failed with NameError on ci (py3.11) while passing locally (py3.14, lazy annotations). add the missing vllm.utils / protocol / SamplingParams symbols to the stub. --- tests/conftest.py | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/tests/conftest.py b/tests/conftest.py index d0d56f4..9c6bef0 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -56,9 +56,33 @@ def _install_vllm_stub(): max_parallel_loading_workers: Optional[int] = None kv_cache_dtype: Optional[str] = None + class _Stub: # pragma: no cover - placeholder for vllm symbols + def __init__(self, *args, **kwargs): + pass + vllm.AsyncEngineArgs = AsyncEngineArgs + vllm.SamplingParams = _Stub sys.modules["vllm"] = vllm + # src.utils imports these at module load and uses ErrorResponse as a return + # annotation, which Python evaluates eagerly on <3.14 -> must be defined. + vllm_utils = types.ModuleType("vllm.utils") + vllm_utils.random_uuid = lambda: "stub-uuid" + vllm.utils = vllm_utils + sys.modules["vllm.utils"] = vllm_utils + + protocol = types.ModuleType("vllm.entrypoints.openai.engine.protocol") + protocol.ErrorResponse = _Stub + protocol.ErrorInfo = _Stub + protocol.RequestResponseMetadata = _Stub + for name in ( + "vllm.entrypoints", + "vllm.entrypoints.openai", + "vllm.entrypoints.openai.engine", + ): + sys.modules.setdefault(name, types.ModuleType(name)) + sys.modules["vllm.entrypoints.openai.engine.protocol"] = protocol + # vllm.model_executor.model_loader.tensorizer.TensorizerConfig model_executor = types.ModuleType("vllm.model_executor") model_loader = types.ModuleType("vllm.model_executor.model_loader") From d7ba3b6ab7ce30c0be175dccf33f1493c968c055 Mon Sep 17 00:00:00 2001 From: Tim Pietrusky Date: Fri, 19 Jun 2026 19:05:24 +0200 Subject: [PATCH 3/3] test: install pyyaml and isolate vllm config file in tests after rebasing onto main, get_engine_args() loads a vllm-style config via PyYAML (a transitive vllm dep). vllm is stubbed in tests, so add pyyaml explicitly and point VLLM_CONFIG_FILE at a nonexistent path so no stray config is picked up. --- tests/requirements.txt | 3 +++ tests/test_engine_args.py | 2 ++ 2 files changed, 5 insertions(+) diff --git a/tests/requirements.txt b/tests/requirements.txt index ce1cb0d..917605c 100644 --- a/tests/requirements.txt +++ b/tests/requirements.txt @@ -1,3 +1,6 @@ # Test-only dependencies. vllm/torch are stubbed in conftest.py when absent, # so the unit tests run on a plain CPU runner without the GPU image. pytest>=8,<10 +# get_engine_args() reads a vLLM-style config via PyYAML (a transitive vllm dep +# at runtime); install it explicitly here since vllm itself is stubbed. +pyyaml diff --git a/tests/test_engine_args.py b/tests/test_engine_args.py index dce4d24..5a85a61 100644 --- a/tests/test_engine_args.py +++ b/tests/test_engine_args.py @@ -94,6 +94,8 @@ class TestGetEngineArgsServedName: # Avoid the network branch in _resolve_max_model_len. monkeypatch.setenv("MAX_NUM_BATCHED_TOKENS", "2048") monkeypatch.delenv("SERVED_MODEL_NAME", raising=False) + # Don't pick up a stray vLLM config file from the environment. + monkeypatch.setenv("VLLM_CONFIG_FILE", "/nonexistent-vllm-config.yaml") @requires_case_sensitive_fs def test_served_name_is_repo_id_when_path_rewritten(self, hf_cache, monkeypatch):