diff --git a/Dockerfile b/Dockerfile index 59b3943..594db86 100644 --- a/Dockerfile +++ b/Dockerfile @@ -8,6 +8,16 @@ ENV PATH="/root/.local/bin:$PATH" RUN ldconfig /usr/local/cuda-13.0/compat/ +# nixl_ep PyPI wheels are compiled against CUDA 12.x and require libcudart.so.12. +# CUDA 13 runtime is ABI-compatible with CUDA 12, so symlinking is safe. +# Symlink into /usr/local/lib so it is in the default linker search path. +RUN ln -sf /usr/local/cuda/lib64/libcudart.so.13 /usr/local/lib/libcudart.so.12 && ldconfig + +# CUDA 13.0 containers return libs to /usr/local/nvidia/lib64 so container +# providers (RunPod, Lambda, etc.) can mount host drivers there consistently. +# See: https://github.com/vllm-project/vllm/issues/18859 +ENV LD_LIBRARY_PATH=/usr/local/nvidia/lib64:/usr/local/cuda/lib64:$LD_LIBRARY_PATH + # Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ uv pip install --system "vllm[flashinfer]==0.21.0" && \ diff --git a/builder/requirements.txt b/builder/requirements.txt index b05ff13..90a2bc2 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -11,5 +11,4 @@ pydantic-settings hf-transfer transformers>=5 bitsandbytes>=0.45.0 -kernels torch-c-dlpack-ext