Merge pull request #272 from runpod-workers/feat/vllm-0.16.0
Release / release (push) Waiting to run

feat: Update to 0.16.0
This commit is contained in:
chrisvela
2026-03-05 13:06:45 -06:00
committed by GitHub
2 changed files with 1 additions and 10 deletions
-9
View File
@@ -280,15 +280,6 @@
"advanced": true
}
},
{
"key": "NUM_GPU_BLOCKS_OVERRIDE",
"input": {
"name": "Num GPU Blocks Override",
"type": "number",
"description": "If specified, ignore GPU profiling result and use this number of GPU blocks.",
"advanced": true
}
},
{
"key": "MAX_NUM_BATCHED_TOKENS",
"input": {
+1 -1
View File
@@ -7,7 +7,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/
# Install vLLM with FlashInfer - use CUDA 12.8 PyTorch wheels (compatible with vLLM 0.15.1)
RUN python3 -m pip install --upgrade pip && \
python3 -m pip install "vllm[flashinfer]==0.15.1" --extra-index-url https://download.pytorch.org/whl/cu129
python3 -m pip install "vllm[flashinfer]==0.16.0" --extra-index-url https://download.pytorch.org/whl/cu129