feat: update 0.16.0, add lmcache
This commit is contained in:
+16
-10
@@ -1,20 +1,21 @@
|
|||||||
FROM nvidia/cuda:12.9.1-base-ubuntu22.04
|
FROM nvidia/cuda:12.9.1-base-ubuntu22.04
|
||||||
|
|
||||||
RUN apt-get update -y \
|
RUN apt-get update -y \
|
||||||
&& apt-get install -y python3-pip
|
&& apt-get install -y python3-pip curl \
|
||||||
|
&& curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||||
|
|
||||||
|
ENV PATH="/root/.local/bin:$PATH"
|
||||||
|
|
||||||
RUN ldconfig /usr/local/cuda-12.9/compat/
|
RUN ldconfig /usr/local/cuda-12.9/compat/
|
||||||
|
|
||||||
# Install vLLM with FlashInfer - use CUDA 12.8 PyTorch wheels (compatible with vLLM 0.15.1)
|
# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels
|
||||||
RUN python3 -m pip install --upgrade pip && \
|
RUN uv pip install --system "packaging>=24.2" && \
|
||||||
python3 -m pip install "vllm[flashinfer]==0.16.0" --extra-index-url https://download.pytorch.org/whl/cu129
|
uv pip install --system "vllm[flashinfer]==0.16.0" --extra-index-url https://download.pytorch.org/whl/cu129
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
||||||
COPY builder/requirements.txt /requirements.txt
|
COPY builder/requirements.txt /requirements.txt
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||||
python3 -m pip install --upgrade -r /requirements.txt
|
uv pip install --system -r /requirements.txt
|
||||||
|
|
||||||
# Setup for Option 2: Building the Image with the Model included
|
# Setup for Option 2: Building the Image with the Model included
|
||||||
ARG MODEL_NAME=""
|
ARG MODEL_NAME=""
|
||||||
@@ -24,6 +25,7 @@ ARG QUANTIZATION=""
|
|||||||
ARG MODEL_REVISION=""
|
ARG MODEL_REVISION=""
|
||||||
ARG TOKENIZER_REVISION=""
|
ARG TOKENIZER_REVISION=""
|
||||||
ARG VLLM_NIGHTLY="false"
|
ARG VLLM_NIGHTLY="false"
|
||||||
|
ARG LMCACHE="false"
|
||||||
|
|
||||||
ENV MODEL_NAME=$MODEL_NAME \
|
ENV MODEL_NAME=$MODEL_NAME \
|
||||||
MODEL_REVISION=$MODEL_REVISION \
|
MODEL_REVISION=$MODEL_REVISION \
|
||||||
@@ -45,10 +47,14 @@ ENV MODEL_NAME=$MODEL_NAME \
|
|||||||
|
|
||||||
ENV PYTHONPATH="/:/vllm-workspace"
|
ENV PYTHONPATH="/:/vllm-workspace"
|
||||||
|
|
||||||
|
RUN if [ "${LMCACHE}" = "true" ]; then \
|
||||||
|
uv pip install --system lmcache; \
|
||||||
|
fi
|
||||||
|
|
||||||
RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \
|
RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \
|
||||||
pip install -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \
|
uv pip install --system -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \
|
||||||
apt-get update && apt-get install -y git && rm -rf /var/lib/apt/lists/* && \
|
apt-get update && apt-get install -y git && rm -rf /var/lib/apt/lists/* && \
|
||||||
pip install git+https://github.com/huggingface/transformers.git; \
|
uv pip install --system git+https://github.com/huggingface/transformers.git; \
|
||||||
fi
|
fi
|
||||||
|
|
||||||
COPY src /src
|
COPY src /src
|
||||||
|
|||||||
@@ -401,6 +401,17 @@ def get_engine_args():
|
|||||||
if os.getenv("MAX_PARALLEL_LOADING_WORKERS"):
|
if os.getenv("MAX_PARALLEL_LOADING_WORKERS"):
|
||||||
logging.warning("Overriding MAX_PARALLEL_LOADING_WORKERS with None because more than 1 GPU is available.")
|
logging.warning("Overriding MAX_PARALLEL_LOADING_WORKERS with None because more than 1 GPU is available.")
|
||||||
|
|
||||||
|
# LMCache requires HMA to be disabled
|
||||||
|
_kv_transfer = args.get("kv_transfer_config")
|
||||||
|
_kv_offload = args.get("kv_offloading_backend")
|
||||||
|
_lmcache_active = _kv_offload == "lmcache" or (
|
||||||
|
isinstance(_kv_transfer, dict)
|
||||||
|
and "lmcache" in str(_kv_transfer.get("kv_connector", "")).lower()
|
||||||
|
)
|
||||||
|
if _lmcache_active and not args.get("disable_hybrid_kv_cache_manager"):
|
||||||
|
args["disable_hybrid_kv_cache_manager"] = True
|
||||||
|
logging.info("LMCache detected: automatically setting disable_hybrid_kv_cache_manager=True")
|
||||||
|
|
||||||
# Deprecated env args backwards compatibility
|
# Deprecated env args backwards compatibility
|
||||||
if args.get("kv_cache_dtype") == "fp8_e5m2":
|
if args.get("kv_cache_dtype") == "fp8_e5m2":
|
||||||
args["kv_cache_dtype"] = "fp8"
|
args["kv_cache_dtype"] = "fp8"
|
||||||
|
|||||||
Reference in New Issue
Block a user