update: update to 0.20.1, update dockerfile to cuda 13

This commit is contained in:
velaraptor-runpod
2026-05-15 11:35:53 -04:00
parent 678bb4be8f
commit 9c139e8ceb
3 changed files with 4 additions and 4 deletions
+1 -1
View File
@@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
[![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
Current vLLM version: [0.20.0](https://github.com/vllm-project/vllm/releases/tag/v0.20.0)
Current vLLM version: [0.20.1](https://github.com/vllm-project/vllm/releases/tag/v0.20.1)
---
+2 -2
View File
@@ -1,4 +1,4 @@
FROM nvidia/cuda:12.9.1-base-ubuntu22.04
FROM nvidia/cuda:13.0.2-base-ubuntu22.04
RUN apt-get update -y \
&& apt-get install -y python3-pip curl \
@@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/
# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels
RUN uv pip install --system "packaging>=24.2" && \
uv pip install --system "vllm[flashinfer]==0.20.1" --extra-index-url https://download.pytorch.org/whl/cu129
uv pip install --system "vllm[flashinfer]==0.20.1" --extra-index-url https://download.pytorch.org/whl/cu130
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
COPY builder/requirements.txt /requirements.txt
+1 -1
View File
@@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png)
Current vLLM version: [0.20.0](https://github.com/vllm-project/vllm/releases/tag/v0.20.0)
Current vLLM version: [0.20.1](https://github.com/vllm-project/vllm/releases/tag/v0.20.1)
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)