diff --git a/.runpod/README.md b/.runpod/README.md index d6ae1cc..50a5674 100644 --- a/.runpod/README.md +++ b/.runpod/README.md @@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API [![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) -Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2) +Current vLLM version: [0.22.1](https://github.com/vllm-project/vllm/releases/tag/v0.22.1) --- diff --git a/Dockerfile b/Dockerfile index 089490c..f6a4245 100644 --- a/Dockerfile +++ b/Dockerfile @@ -21,7 +21,7 @@ ENV LD_LIBRARY_PATH=/usr/local/nvidia/lib64:/usr/local/cuda/lib64:$LD_LIBRARY_PA # Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.21.0" && \ + uv pip install --system "vllm[flashinfer]==0.22.1" && \ uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) diff --git a/README.md b/README.md index 372be50..16ff521 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https: ![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png) -Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2) +Current vLLM version: [0.22.1](https://github.com/vllm-project/vllm/releases/tag/v0.22.1) > Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)