feat: Update to 0.16.0, remove NUM_GPU_BLOCKS_OVERRIDE in hub default since 0 will break

This commit is contained in:
velaraptor-runpod
2026-03-04 16:38:40 -06:00
parent 13fa71878e
commit 2b5f07df63
2 changed files with 1 additions and 10 deletions
-9
View File
@@ -280,15 +280,6 @@
"advanced": true "advanced": true
} }
}, },
{
"key": "NUM_GPU_BLOCKS_OVERRIDE",
"input": {
"name": "Num GPU Blocks Override",
"type": "number",
"description": "If specified, ignore GPU profiling result and use this number of GPU blocks.",
"advanced": true
}
},
{ {
"key": "MAX_NUM_BATCHED_TOKENS", "key": "MAX_NUM_BATCHED_TOKENS",
"input": { "input": {
+1 -1
View File
@@ -7,7 +7,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/
# Install vLLM with FlashInfer - use CUDA 12.8 PyTorch wheels (compatible with vLLM 0.15.1) # Install vLLM with FlashInfer - use CUDA 12.8 PyTorch wheels (compatible with vLLM 0.15.1)
RUN python3 -m pip install --upgrade pip && \ RUN python3 -m pip install --upgrade pip && \
python3 -m pip install "vllm[flashinfer]==0.15.1" --extra-index-url https://download.pytorch.org/whl/cu129 python3 -m pip install "vllm[flashinfer]==0.16.0" --extra-index-url https://download.pytorch.org/whl/cu129