Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9e1c483136 | ||
|
|
d71ea9939d | ||
|
|
7e2b4e2288 | ||
|
|
015f8f3c4d | ||
|
|
75ffcf73f2 | ||
|
|
d7ba3b6ab7 | ||
|
|
b11c91722c | ||
|
|
fcdc799e0d | ||
|
|
84ec446493 | ||
|
|
db246653a2 | ||
|
|
1b3228a2dc | ||
|
|
4817d4a8e7 | ||
|
|
0378382a92 | ||
|
|
08580e7ccf | ||
|
|
8868aae6b1 | ||
|
|
fb8adc5c06 |
@@ -0,0 +1,32 @@
|
|||||||
|
name: Tests
|
||||||
|
|
||||||
|
on:
|
||||||
|
pull_request:
|
||||||
|
branches:
|
||||||
|
- "**"
|
||||||
|
push:
|
||||||
|
branches:
|
||||||
|
- "main"
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
pytest:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Set up Python
|
||||||
|
uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: "3.11"
|
||||||
|
|
||||||
|
- name: Install test dependencies
|
||||||
|
run: |
|
||||||
|
python -m pip install --upgrade pip
|
||||||
|
pip install -r tests/requirements.txt
|
||||||
|
|
||||||
|
- name: Run unit tests
|
||||||
|
run: python -m pytest tests -v
|
||||||
+1
-1
@@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
|
|||||||
|
|
||||||
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
||||||
|
|
||||||
Current vLLM version: [0.22.1](https://github.com/vllm-project/vllm/releases/tag/v0.22.1)
|
Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -30,7 +30,7 @@
|
|||||||
}
|
}
|
||||||
],
|
],
|
||||||
"config": {
|
"config": {
|
||||||
"gpuTypeId": "NNVIDIA L40",
|
"gpuTypeId": "NVIDIA L40",
|
||||||
"gpuCount": 1,
|
"gpuCount": 1,
|
||||||
"env": [
|
"env": [
|
||||||
{
|
{
|
||||||
|
|||||||
+1
-12
@@ -8,20 +8,9 @@ ENV PATH="/root/.local/bin:$PATH"
|
|||||||
|
|
||||||
RUN ldconfig /usr/local/cuda-13.0/compat/
|
RUN ldconfig /usr/local/cuda-13.0/compat/
|
||||||
|
|
||||||
# nixl_ep PyPI wheels are compiled against CUDA 12.x and require libcudart.so.12.
|
|
||||||
# CUDA 13 runtime is ABI-compatible with CUDA 12, so symlinking is safe.
|
|
||||||
# Symlink into /usr/local/cuda/lib64 (already in LD_LIBRARY_PATH) so the linker
|
|
||||||
# finds it by filename scan rather than relying on ldcache SONAME lookup.
|
|
||||||
RUN ln -sf /usr/local/cuda/lib64/libcudart.so.13 /usr/local/cuda/lib64/libcudart.so.12 && ldconfig
|
|
||||||
|
|
||||||
# CUDA 13.0 containers return libs to /usr/local/nvidia/lib64 so container
|
|
||||||
# providers (RunPod, Lambda, etc.) can mount host drivers there consistently.
|
|
||||||
# See: https://github.com/vllm-project/vllm/issues/18859
|
|
||||||
ENV LD_LIBRARY_PATH=/usr/local/nvidia/lib64:/usr/local/cuda/lib64:$LD_LIBRARY_PATH
|
|
||||||
|
|
||||||
# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
|
# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
|
||||||
RUN uv pip install --system "packaging>=24.2" && \
|
RUN uv pip install --system "packaging>=24.2" && \
|
||||||
uv pip install --system "vllm[flashinfer]==0.22.1" && \
|
uv pip install --system "vllm[flashinfer]==0.20.2" && \
|
||||||
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation
|
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation
|
||||||
|
|
||||||
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
|
|||||||
|
|
||||||

|

|
||||||
|
|
||||||
Current vLLM version: [0.22.1](https://github.com/vllm-project/vllm/releases/tag/v0.22.1)
|
Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
|
||||||
|
|
||||||
|
|
||||||
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
|
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
|
||||||
|
|||||||
@@ -1,9 +1,9 @@
|
|||||||
ray
|
ray
|
||||||
pandas
|
pandas
|
||||||
pyarrow
|
pyarrow
|
||||||
runpod==1.9.1
|
runpod~=1.10.0
|
||||||
huggingface-hub
|
huggingface-hub
|
||||||
lmcache==0.4.6
|
lmcache==0.4.5
|
||||||
packaging>=24.2
|
packaging>=24.2
|
||||||
typing-extensions>=4.8.0
|
typing-extensions>=4.8.0
|
||||||
pydantic
|
pydantic
|
||||||
|
|||||||
@@ -0,0 +1,11 @@
|
|||||||
|
model: google/gemma-4-31b-it
|
||||||
|
gpu-memory-utilization: 0.95
|
||||||
|
max-model-len: 8192
|
||||||
|
dtype: auto
|
||||||
|
trust-remote-code: true
|
||||||
|
quantization: fp8
|
||||||
|
kv-cache-dtype: fp8
|
||||||
|
enforce-eager: false
|
||||||
|
enable-prefix-caching: true
|
||||||
|
enable-chunked-prefill: true
|
||||||
|
speculative-config: '{"model":"RedHatAI/gemma-4-31B-it-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||||
@@ -0,0 +1,9 @@
|
|||||||
|
model: openai/gpt-oss-120b
|
||||||
|
gpu-memory-utilization: 0.95
|
||||||
|
max-model-len: 8192
|
||||||
|
dtype: auto
|
||||||
|
trust-remote-code: true
|
||||||
|
enforce-eager: false
|
||||||
|
enable-prefix-caching: true
|
||||||
|
enable-chunked-prefill: true
|
||||||
|
speculative-config: '{"model":"RedHatAI/gpt-oss-120b-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||||
+8
-1
@@ -601,6 +601,13 @@ def get_engine_args():
|
|||||||
|
|
||||||
# Resolve lowercase HF cache paths (FDE-174)
|
# Resolve lowercase HF cache paths (FDE-174)
|
||||||
if args.get("model"):
|
if args.get("model"):
|
||||||
args["model"] = _resolve_cached_model_path(args["model"])
|
original_model = args["model"]
|
||||||
|
args["model"] = _resolve_cached_model_path(original_model)
|
||||||
|
# When the model was rewritten to an on-disk snapshot path, keep serving
|
||||||
|
# under the original repo id so the OpenAI API model name does not become
|
||||||
|
# a filesystem path (issue #310). An explicit served_model_name (or the
|
||||||
|
# OPENAI_SERVED_MODEL_NAME_OVERRIDE handled downstream) still wins.
|
||||||
|
if args["model"] != original_model and not args.get("served_model_name"):
|
||||||
|
args["served_model_name"] = original_model
|
||||||
|
|
||||||
return AsyncEngineArgs(**args)
|
return AsyncEngineArgs(**args)
|
||||||
|
|||||||
@@ -0,0 +1,105 @@
|
|||||||
|
"""Shared test fixtures.
|
||||||
|
|
||||||
|
``src/engine_args.py`` hard-imports ``vllm`` (and a tensorizer submodule) and
|
||||||
|
``torch.cuda``. Both are only installed inside the GPU Docker image, so when the
|
||||||
|
tests run on a machine without them we install lightweight stubs. When the real
|
||||||
|
packages *are* available (e.g. CI inside the worker image) the stubs are skipped
|
||||||
|
and the real ones are used instead.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import sys
|
||||||
|
import types
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from typing import Optional, Union, List
|
||||||
|
|
||||||
|
|
||||||
|
def _install_torch_stub():
|
||||||
|
try:
|
||||||
|
import torch # noqa: F401
|
||||||
|
return # real torch present, nothing to stub
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
torch = types.ModuleType("torch")
|
||||||
|
cuda = types.ModuleType("torch.cuda")
|
||||||
|
# No GPU in the test environment -> 0 devices (skips tensor-parallel setup).
|
||||||
|
cuda.device_count = lambda: 0
|
||||||
|
torch.cuda = cuda
|
||||||
|
sys.modules["torch"] = torch
|
||||||
|
sys.modules["torch.cuda"] = cuda
|
||||||
|
|
||||||
|
|
||||||
|
def _install_vllm_stub():
|
||||||
|
try:
|
||||||
|
import vllm # noqa: F401
|
||||||
|
return # real vLLM present, nothing to stub
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
vllm = types.ModuleType("vllm")
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class AsyncEngineArgs:
|
||||||
|
# Only the fields the worker actually sets/reads need to exist here;
|
||||||
|
# get_engine_args() filters args down to AsyncEngineArgs.__dataclass_fields__
|
||||||
|
# before construction, so unknown keys are dropped rather than passed.
|
||||||
|
model: Optional[str] = None
|
||||||
|
served_model_name: Optional[Union[str, List[str]]] = None
|
||||||
|
revision: Optional[str] = None
|
||||||
|
tokenizer: Optional[str] = None
|
||||||
|
trust_remote_code: bool = False
|
||||||
|
max_model_len: Optional[int] = None
|
||||||
|
max_num_batched_tokens: Optional[int] = None
|
||||||
|
disable_log_stats: bool = False
|
||||||
|
gpu_memory_utilization: float = 0.9
|
||||||
|
tensor_parallel_size: int = 1
|
||||||
|
max_parallel_loading_workers: Optional[int] = None
|
||||||
|
kv_cache_dtype: Optional[str] = None
|
||||||
|
|
||||||
|
class _Stub: # pragma: no cover - placeholder for vllm symbols
|
||||||
|
def __init__(self, *args, **kwargs):
|
||||||
|
pass
|
||||||
|
|
||||||
|
vllm.AsyncEngineArgs = AsyncEngineArgs
|
||||||
|
vllm.SamplingParams = _Stub
|
||||||
|
sys.modules["vllm"] = vllm
|
||||||
|
|
||||||
|
# src.utils imports these at module load and uses ErrorResponse as a return
|
||||||
|
# annotation, which Python evaluates eagerly on <3.14 -> must be defined.
|
||||||
|
vllm_utils = types.ModuleType("vllm.utils")
|
||||||
|
vllm_utils.random_uuid = lambda: "stub-uuid"
|
||||||
|
vllm.utils = vllm_utils
|
||||||
|
sys.modules["vllm.utils"] = vllm_utils
|
||||||
|
|
||||||
|
protocol = types.ModuleType("vllm.entrypoints.openai.engine.protocol")
|
||||||
|
protocol.ErrorResponse = _Stub
|
||||||
|
protocol.ErrorInfo = _Stub
|
||||||
|
protocol.RequestResponseMetadata = _Stub
|
||||||
|
for name in (
|
||||||
|
"vllm.entrypoints",
|
||||||
|
"vllm.entrypoints.openai",
|
||||||
|
"vllm.entrypoints.openai.engine",
|
||||||
|
):
|
||||||
|
sys.modules.setdefault(name, types.ModuleType(name))
|
||||||
|
sys.modules["vllm.entrypoints.openai.engine.protocol"] = protocol
|
||||||
|
|
||||||
|
# vllm.model_executor.model_loader.tensorizer.TensorizerConfig
|
||||||
|
model_executor = types.ModuleType("vllm.model_executor")
|
||||||
|
model_loader = types.ModuleType("vllm.model_executor.model_loader")
|
||||||
|
tensorizer = types.ModuleType("vllm.model_executor.model_loader.tensorizer")
|
||||||
|
|
||||||
|
class TensorizerConfig: # pragma: no cover - placeholder
|
||||||
|
def __init__(self, *args, **kwargs):
|
||||||
|
pass
|
||||||
|
|
||||||
|
tensorizer.TensorizerConfig = TensorizerConfig
|
||||||
|
model_loader.tensorizer = tensorizer
|
||||||
|
model_executor.model_loader = model_loader
|
||||||
|
vllm.model_executor = model_executor
|
||||||
|
sys.modules["vllm.model_executor"] = model_executor
|
||||||
|
sys.modules["vllm.model_executor.model_loader"] = model_loader
|
||||||
|
sys.modules["vllm.model_executor.model_loader.tensorizer"] = tensorizer
|
||||||
|
|
||||||
|
|
||||||
|
_install_torch_stub()
|
||||||
|
_install_vllm_stub()
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
# Test-only dependencies. vllm/torch are stubbed in conftest.py when absent,
|
||||||
|
# so the unit tests run on a plain CPU runner without the GPU image.
|
||||||
|
pytest>=8,<10
|
||||||
|
# get_engine_args() reads a vLLM-style config via PyYAML (a transitive vllm dep
|
||||||
|
# at runtime); install it explicitly here since vllm itself is stubbed.
|
||||||
|
pyyaml
|
||||||
@@ -0,0 +1,126 @@
|
|||||||
|
"""Tests for HF cache path resolution and served-model-name decoupling.
|
||||||
|
|
||||||
|
Regression coverage for issue #310: when MODEL_NAME is served from a lowercased
|
||||||
|
HF cache dir, the cache resolver rewrites engine_args.model to a snapshot path.
|
||||||
|
The served model name must stay the original repo id, not the path.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from src import engine_args
|
||||||
|
from src.engine_args import _resolve_cached_model_path, get_engine_args
|
||||||
|
|
||||||
|
|
||||||
|
MODEL = "Qwen/Qwen3.6-27B-FP8"
|
||||||
|
SNAPSHOT_HASH = "e89b16ebf1988b3d6befa7de50abc2d76f26eb09"
|
||||||
|
|
||||||
|
|
||||||
|
def _make_cache(root, folder_name, snapshot=SNAPSHOT_HASH):
|
||||||
|
"""Create a HF-style ``models--…/snapshots/<hash>/`` dir and return its path."""
|
||||||
|
snap_dir = os.path.join(root, folder_name, "snapshots", snapshot)
|
||||||
|
os.makedirs(snap_dir)
|
||||||
|
return snap_dir
|
||||||
|
|
||||||
|
|
||||||
|
def _is_case_sensitive_fs(path):
|
||||||
|
"""The lowercase-cache resolution only matters on case-sensitive filesystems.
|
||||||
|
|
||||||
|
On macOS (APFS, case-insensitive by default) ``models--Qwen--…`` and
|
||||||
|
``models--qwen--…`` collide, so the resolver always sees the exact-case dir
|
||||||
|
as present. Production runs on Linux (case-sensitive), which is what these
|
||||||
|
tests exercise.
|
||||||
|
"""
|
||||||
|
probe = os.path.join(path, "CaseProbe")
|
||||||
|
open(probe, "w").close()
|
||||||
|
try:
|
||||||
|
return not os.path.exists(os.path.join(path, "caseprobe"))
|
||||||
|
finally:
|
||||||
|
os.remove(probe)
|
||||||
|
|
||||||
|
|
||||||
|
requires_case_sensitive_fs = pytest.mark.skipif(
|
||||||
|
not _is_case_sensitive_fs(os.environ.get("TMPDIR", "/tmp")),
|
||||||
|
reason="lowercase HF cache resolution only applies on case-sensitive filesystems",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def hf_cache(tmp_path, monkeypatch):
|
||||||
|
cache = tmp_path / "hub"
|
||||||
|
cache.mkdir()
|
||||||
|
monkeypatch.setenv("HUGGINGFACE_HUB_CACHE", str(cache))
|
||||||
|
# Make sure HF_HOME does not shadow the explicit cache dir during the test.
|
||||||
|
monkeypatch.delenv("HF_HOME", raising=False)
|
||||||
|
return cache
|
||||||
|
|
||||||
|
|
||||||
|
class TestResolveCachedModelPath:
|
||||||
|
def test_exact_case_dir_returns_repo_id(self, hf_cache):
|
||||||
|
_make_cache(str(hf_cache), "models--Qwen--Qwen3.6-27B-FP8")
|
||||||
|
assert _resolve_cached_model_path(MODEL) == MODEL
|
||||||
|
|
||||||
|
def test_no_cache_returns_repo_id(self, hf_cache):
|
||||||
|
assert _resolve_cached_model_path(MODEL) == MODEL
|
||||||
|
|
||||||
|
def test_absolute_path_passthrough(self, hf_cache):
|
||||||
|
path = "/runpod-volume/some/local/model"
|
||||||
|
assert _resolve_cached_model_path(path) == path
|
||||||
|
|
||||||
|
@requires_case_sensitive_fs
|
||||||
|
def test_lowercase_dir_returns_snapshot_path(self, hf_cache):
|
||||||
|
snap = _make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")
|
||||||
|
assert _resolve_cached_model_path(MODEL) == snap
|
||||||
|
|
||||||
|
def test_lowercase_dir_without_snapshots_returns_repo_id(self, hf_cache):
|
||||||
|
# Dir exists but has no snapshots subdir -> nothing to resolve to.
|
||||||
|
os.makedirs(os.path.join(str(hf_cache), "models--qwen--qwen3.6-27b-fp8"))
|
||||||
|
assert _resolve_cached_model_path(MODEL) == MODEL
|
||||||
|
|
||||||
|
@requires_case_sensitive_fs
|
||||||
|
def test_lowercase_dir_picks_latest_snapshot(self, hf_cache):
|
||||||
|
folder = "models--qwen--qwen3.6-27b-fp8"
|
||||||
|
_make_cache(str(hf_cache), folder, snapshot="aaaa")
|
||||||
|
latest = _make_cache(str(hf_cache), folder, snapshot="zzzz")
|
||||||
|
assert _resolve_cached_model_path(MODEL) == latest
|
||||||
|
|
||||||
|
|
||||||
|
class TestGetEngineArgsServedName:
|
||||||
|
"""Issue #310: served name must be decoupled from the resolved on-disk path."""
|
||||||
|
|
||||||
|
@pytest.fixture(autouse=True)
|
||||||
|
def base_env(self, monkeypatch):
|
||||||
|
# Avoid the network branch in _resolve_max_model_len.
|
||||||
|
monkeypatch.setenv("MAX_NUM_BATCHED_TOKENS", "2048")
|
||||||
|
monkeypatch.delenv("SERVED_MODEL_NAME", raising=False)
|
||||||
|
# Don't pick up a stray vLLM config file from the environment.
|
||||||
|
monkeypatch.setenv("VLLM_CONFIG_FILE", "/nonexistent-vllm-config.yaml")
|
||||||
|
|
||||||
|
@requires_case_sensitive_fs
|
||||||
|
def test_served_name_is_repo_id_when_path_rewritten(self, hf_cache, monkeypatch):
|
||||||
|
snap = _make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")
|
||||||
|
monkeypatch.setenv("MODEL_NAME", MODEL)
|
||||||
|
|
||||||
|
result = get_engine_args()
|
||||||
|
|
||||||
|
assert result.model == snap # weights load from the lowercase cache
|
||||||
|
assert result.served_model_name == MODEL # API still serves the repo id
|
||||||
|
|
||||||
|
def test_served_name_untouched_when_no_rewrite(self, hf_cache, monkeypatch):
|
||||||
|
_make_cache(str(hf_cache), "models--Qwen--Qwen3.6-27B-FP8")
|
||||||
|
monkeypatch.setenv("MODEL_NAME", MODEL)
|
||||||
|
|
||||||
|
result = get_engine_args()
|
||||||
|
|
||||||
|
assert result.model == MODEL
|
||||||
|
assert result.served_model_name is None
|
||||||
|
|
||||||
|
def test_explicit_served_name_not_overridden(self, hf_cache, monkeypatch):
|
||||||
|
_make_cache(str(hf_cache), "models--qwen--qwen3.6-27b-fp8")
|
||||||
|
monkeypatch.setenv("MODEL_NAME", MODEL)
|
||||||
|
monkeypatch.setenv("SERVED_MODEL_NAME", "custom-name")
|
||||||
|
|
||||||
|
result = get_engine_args()
|
||||||
|
|
||||||
|
assert result.served_model_name == "custom-name"
|
||||||
Reference in New Issue
Block a user