From cb3f077dbabb745aae108281037d90d10df0a1f2 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Wed, 3 Jun 2026 14:43:56 -0500 Subject: [PATCH 1/3] feat: upgrade vllm to 0.21.0 --- Dockerfile | 2 +- builder/requirements.txt | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/Dockerfile b/Dockerfile index 5e4efd1..59b3943 100644 --- a/Dockerfile +++ b/Dockerfile @@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-13.0/compat/ # Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.20.2" && \ + uv pip install --system "vllm[flashinfer]==0.21.0" && \ uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) diff --git a/builder/requirements.txt b/builder/requirements.txt index 19fceca..b05ff13 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -3,7 +3,7 @@ pandas pyarrow runpod==1.9.0 huggingface-hub -lmcache==0.4.5 +lmcache==0.4.6 packaging>=24.2 typing-extensions>=4.8.0 pydantic From 9618e799ba055c664ffd21ba6ca47f72b4df22c8 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 4 Jun 2026 15:35:58 -0500 Subject: [PATCH 2/3] chore: fix cuda libraries --- Dockerfile | 10 ++++++++++ builder/requirements.txt | 1 - 2 files changed, 10 insertions(+), 1 deletion(-) diff --git a/Dockerfile b/Dockerfile index 59b3943..594db86 100644 --- a/Dockerfile +++ b/Dockerfile @@ -8,6 +8,16 @@ ENV PATH="/root/.local/bin:$PATH" RUN ldconfig /usr/local/cuda-13.0/compat/ +# nixl_ep PyPI wheels are compiled against CUDA 12.x and require libcudart.so.12. +# CUDA 13 runtime is ABI-compatible with CUDA 12, so symlinking is safe. +# Symlink into /usr/local/lib so it is in the default linker search path. +RUN ln -sf /usr/local/cuda/lib64/libcudart.so.13 /usr/local/lib/libcudart.so.12 && ldconfig + +# CUDA 13.0 containers return libs to /usr/local/nvidia/lib64 so container +# providers (RunPod, Lambda, etc.) can mount host drivers there consistently. +# See: https://github.com/vllm-project/vllm/issues/18859 +ENV LD_LIBRARY_PATH=/usr/local/nvidia/lib64:/usr/local/cuda/lib64:$LD_LIBRARY_PATH + # Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ uv pip install --system "vllm[flashinfer]==0.21.0" && \ diff --git a/builder/requirements.txt b/builder/requirements.txt index b05ff13..90a2bc2 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -11,5 +11,4 @@ pydantic-settings hf-transfer transformers>=5 bitsandbytes>=0.45.0 -kernels torch-c-dlpack-ext From c8ce53c72c10eae0c3f535a19346a175602a5791 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Wed, 10 Jun 2026 16:01:53 -0500 Subject: [PATCH 3/3] fix: add kenels, and fix for cuda --- Dockerfile | 5 +++-- builder/requirements.txt | 1 + 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/Dockerfile b/Dockerfile index 594db86..089490c 100644 --- a/Dockerfile +++ b/Dockerfile @@ -10,8 +10,9 @@ RUN ldconfig /usr/local/cuda-13.0/compat/ # nixl_ep PyPI wheels are compiled against CUDA 12.x and require libcudart.so.12. # CUDA 13 runtime is ABI-compatible with CUDA 12, so symlinking is safe. -# Symlink into /usr/local/lib so it is in the default linker search path. -RUN ln -sf /usr/local/cuda/lib64/libcudart.so.13 /usr/local/lib/libcudart.so.12 && ldconfig +# Symlink into /usr/local/cuda/lib64 (already in LD_LIBRARY_PATH) so the linker +# finds it by filename scan rather than relying on ldcache SONAME lookup. +RUN ln -sf /usr/local/cuda/lib64/libcudart.so.13 /usr/local/cuda/lib64/libcudart.so.12 && ldconfig # CUDA 13.0 containers return libs to /usr/local/nvidia/lib64 so container # providers (RunPod, Lambda, etc.) can mount host drivers there consistently. diff --git a/builder/requirements.txt b/builder/requirements.txt index 90a2bc2..5603b1b 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -11,4 +11,5 @@ pydantic-settings hf-transfer transformers>=5 bitsandbytes>=0.45.0 +kernels<0.15 torch-c-dlpack-ext