feat: Update to 0.16.0, remove NUM_GPU_BLOCKS_OVERRIDE in hub default since 0 will break
This commit is contained in:
+1
-1
@@ -7,7 +7,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/
|
||||
|
||||
# Install vLLM with FlashInfer - use CUDA 12.8 PyTorch wheels (compatible with vLLM 0.15.1)
|
||||
RUN python3 -m pip install --upgrade pip && \
|
||||
python3 -m pip install "vllm[flashinfer]==0.15.1" --extra-index-url https://download.pytorch.org/whl/cu129
|
||||
python3 -m pip install "vllm[flashinfer]==0.16.0" --extra-index-url https://download.pytorch.org/whl/cu129
|
||||
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user