diff --git a/.runpod/README.md b/.runpod/README.md index 18e4369..390d4c2 100644 --- a/.runpod/README.md +++ b/.runpod/README.md @@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API [![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) -Current vLLM version: [0.20.0](https://github.com/vllm-project/vllm/releases/tag/v0.20.0) +Current vLLM version: [0.20.1](https://github.com/vllm-project/vllm/releases/tag/v0.20.1) --- diff --git a/Dockerfile b/Dockerfile index 66227cb..7005291 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,4 +1,4 @@ -FROM nvidia/cuda:12.9.1-base-ubuntu22.04 +FROM nvidia/cuda:13.0.2-base-ubuntu22.04 RUN apt-get update -y \ && apt-get install -y python3-pip curl \ @@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/ # Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.20.1" --extra-index-url https://download.pytorch.org/whl/cu129 + uv pip install --system "vllm[flashinfer]==0.20.1" --extra-index-url https://download.pytorch.org/whl/cu130 # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) COPY builder/requirements.txt /requirements.txt diff --git a/README.md b/README.md index 42dd66d..e25a02f 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https: ![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png) -Current vLLM version: [0.20.0](https://github.com/vllm-project/vllm/releases/tag/v0.20.0) +Current vLLM version: [0.20.1](https://github.com/vllm-project/vllm/releases/tag/v0.20.1) > Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)