feat: Update to 0.16.0, remove NUM_GPU_BLOCKS_OVERRIDE in hub default since 0 will break

This commit is contained in:
velaraptor-runpod
2026-03-04 16:38:40 -06:00
parent 13fa71878e
commit 2b5f07df63
2 changed files with 1 additions and 10 deletions
+1 -1
View File
@@ -7,7 +7,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/
# Install vLLM with FlashInfer - use CUDA 12.8 PyTorch wheels (compatible with vLLM 0.15.1)
RUN python3 -m pip install --upgrade pip && \
python3 -m pip install "vllm[flashinfer]==0.15.1" --extra-index-url https://download.pytorch.org/whl/cu129
python3 -m pip install "vllm[flashinfer]==0.16.0" --extra-index-url https://download.pytorch.org/whl/cu129