From 4817d4a8e7a74e5702ab42d325f5eac4be210251 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Fri, 12 Jun 2026 15:49:23 -0500 Subject: [PATCH] revert to 0.20.2 --- Dockerfile | 13 +------------ builder/requirements.txt | 2 +- 2 files changed, 2 insertions(+), 13 deletions(-) diff --git a/Dockerfile b/Dockerfile index f6a4245..5e4efd1 100644 --- a/Dockerfile +++ b/Dockerfile @@ -8,20 +8,9 @@ ENV PATH="/root/.local/bin:$PATH" RUN ldconfig /usr/local/cuda-13.0/compat/ -# nixl_ep PyPI wheels are compiled against CUDA 12.x and require libcudart.so.12. -# CUDA 13 runtime is ABI-compatible with CUDA 12, so symlinking is safe. -# Symlink into /usr/local/cuda/lib64 (already in LD_LIBRARY_PATH) so the linker -# finds it by filename scan rather than relying on ldcache SONAME lookup. -RUN ln -sf /usr/local/cuda/lib64/libcudart.so.13 /usr/local/cuda/lib64/libcudart.so.12 && ldconfig - -# CUDA 13.0 containers return libs to /usr/local/nvidia/lib64 so container -# providers (RunPod, Lambda, etc.) can mount host drivers there consistently. -# See: https://github.com/vllm-project/vllm/issues/18859 -ENV LD_LIBRARY_PATH=/usr/local/nvidia/lib64:/usr/local/cuda/lib64:$LD_LIBRARY_PATH - # Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.22.1" && \ + uv pip install --system "vllm[flashinfer]==0.20.2" && \ uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) diff --git a/builder/requirements.txt b/builder/requirements.txt index cd0559b..042dbda 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -3,7 +3,7 @@ pandas pyarrow runpod==1.9.1 huggingface-hub -lmcache==0.4.6 +lmcache==0.4.5 packaging>=24.2 typing-extensions>=4.8.0 pydantic