Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
17efb0e7d0 | ||
|
|
2b5f07df63 |
@@ -280,15 +280,6 @@
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "NUM_GPU_BLOCKS_OVERRIDE",
|
||||
"input": {
|
||||
"name": "Num GPU Blocks Override",
|
||||
"type": "number",
|
||||
"description": "If specified, ignore GPU profiling result and use this number of GPU blocks.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "MAX_NUM_BATCHED_TOKENS",
|
||||
"input": {
|
||||
|
||||
+1
-1
@@ -7,7 +7,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/
|
||||
|
||||
# Install vLLM with FlashInfer - use CUDA 12.8 PyTorch wheels (compatible with vLLM 0.15.1)
|
||||
RUN python3 -m pip install --upgrade pip && \
|
||||
python3 -m pip install "vllm[flashinfer]==0.15.1" --extra-index-url https://download.pytorch.org/whl/cu129
|
||||
python3 -m pip install "vllm[flashinfer]==0.16.0" --extra-index-url https://download.pytorch.org/whl/cu129
|
||||
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user