From d8ed3b53534d11c5c915d5494b4d050dbd4621f3 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Tue, 17 Mar 2026 19:54:11 -0500 Subject: [PATCH] feat: update 0.16.0, add lmcache --- Dockerfile | 26 ++++++++++++++++---------- src/engine_args.py | 11 +++++++++++ 2 files changed, 27 insertions(+), 10 deletions(-) diff --git a/Dockerfile b/Dockerfile index 6bb5c46..bcf4069 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,20 +1,21 @@ FROM nvidia/cuda:12.9.1-base-ubuntu22.04 RUN apt-get update -y \ - && apt-get install -y python3-pip + && apt-get install -y python3-pip curl \ + && curl -LsSf https://astral.sh/uv/install.sh | sh + +ENV PATH="/root/.local/bin:$PATH" RUN ldconfig /usr/local/cuda-12.9/compat/ -# Install vLLM with FlashInfer - use CUDA 12.8 PyTorch wheels (compatible with vLLM 0.15.1) -RUN python3 -m pip install --upgrade pip && \ - python3 -m pip install "vllm[flashinfer]==0.16.0" --extra-index-url https://download.pytorch.org/whl/cu129 - - +# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels +RUN uv pip install --system "packaging>=24.2" && \ + uv pip install --system "vllm[flashinfer]==0.16.0" --extra-index-url https://download.pytorch.org/whl/cu129 # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) COPY builder/requirements.txt /requirements.txt -RUN --mount=type=cache,target=/root/.cache/pip \ - python3 -m pip install --upgrade -r /requirements.txt +RUN --mount=type=cache,target=/root/.cache/uv \ + uv pip install --system -r /requirements.txt # Setup for Option 2: Building the Image with the Model included ARG MODEL_NAME="" @@ -24,6 +25,7 @@ ARG QUANTIZATION="" ARG MODEL_REVISION="" ARG TOKENIZER_REVISION="" ARG VLLM_NIGHTLY="false" +ARG LMCACHE="false" ENV MODEL_NAME=$MODEL_NAME \ MODEL_REVISION=$MODEL_REVISION \ @@ -45,10 +47,14 @@ ENV MODEL_NAME=$MODEL_NAME \ ENV PYTHONPATH="/:/vllm-workspace" +RUN if [ "${LMCACHE}" = "true" ]; then \ + uv pip install --system lmcache; \ +fi + RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \ - pip install -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \ + uv pip install --system -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \ apt-get update && apt-get install -y git && rm -rf /var/lib/apt/lists/* && \ - pip install git+https://github.com/huggingface/transformers.git; \ + uv pip install --system git+https://github.com/huggingface/transformers.git; \ fi COPY src /src diff --git a/src/engine_args.py b/src/engine_args.py index 5ae676b..05dd2b7 100644 --- a/src/engine_args.py +++ b/src/engine_args.py @@ -401,6 +401,17 @@ def get_engine_args(): if os.getenv("MAX_PARALLEL_LOADING_WORKERS"): logging.warning("Overriding MAX_PARALLEL_LOADING_WORKERS with None because more than 1 GPU is available.") + # LMCache requires HMA to be disabled + _kv_transfer = args.get("kv_transfer_config") + _kv_offload = args.get("kv_offloading_backend") + _lmcache_active = _kv_offload == "lmcache" or ( + isinstance(_kv_transfer, dict) + and "lmcache" in str(_kv_transfer.get("kv_connector", "")).lower() + ) + if _lmcache_active and not args.get("disable_hybrid_kv_cache_manager"): + args["disable_hybrid_kv_cache_manager"] = True + logging.info("LMCache detected: automatically setting disable_hybrid_kv_cache_manager=True") + # Deprecated env args backwards compatibility if args.get("kv_cache_dtype") == "fp8_e5m2": args["kv_cache_dtype"] = "fp8"