diff --git a/Dockerfile b/Dockerfile index 5e4efd1..59b3943 100644 --- a/Dockerfile +++ b/Dockerfile @@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-13.0/compat/ # Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.20.2" && \ + uv pip install --system "vllm[flashinfer]==0.21.0" && \ uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) diff --git a/builder/requirements.txt b/builder/requirements.txt index 19fceca..b05ff13 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -3,7 +3,7 @@ pandas pyarrow runpod==1.9.0 huggingface-hub -lmcache==0.4.5 +lmcache==0.4.6 packaging>=24.2 typing-extensions>=4.8.0 pydantic