update: update to 0.20.1, update dockerfile to cuda 13
This commit is contained in:
+1
-1
@@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
|
||||
|
||||
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
||||
|
||||
Current vLLM version: [0.20.0](https://github.com/vllm-project/vllm/releases/tag/v0.20.0)
|
||||
Current vLLM version: [0.20.1](https://github.com/vllm-project/vllm/releases/tag/v0.20.1)
|
||||
|
||||
---
|
||||
|
||||
|
||||
+2
-2
@@ -1,4 +1,4 @@
|
||||
FROM nvidia/cuda:12.9.1-base-ubuntu22.04
|
||||
FROM nvidia/cuda:13.0.2-base-ubuntu22.04
|
||||
|
||||
RUN apt-get update -y \
|
||||
&& apt-get install -y python3-pip curl \
|
||||
@@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/
|
||||
|
||||
# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels
|
||||
RUN uv pip install --system "packaging>=24.2" && \
|
||||
uv pip install --system "vllm[flashinfer]==0.20.1" --extra-index-url https://download.pytorch.org/whl/cu129
|
||||
uv pip install --system "vllm[flashinfer]==0.20.1" --extra-index-url https://download.pytorch.org/whl/cu130
|
||||
|
||||
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
||||
COPY builder/requirements.txt /requirements.txt
|
||||
|
||||
@@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
|
||||
|
||||

|
||||
|
||||
Current vLLM version: [0.20.0](https://github.com/vllm-project/vllm/releases/tag/v0.20.0)
|
||||
Current vLLM version: [0.20.1](https://github.com/vllm-project/vllm/releases/tag/v0.20.1)
|
||||
|
||||
|
||||
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
|
||||
|
||||
Reference in New Issue
Block a user