From 9de17d49b72e9a4e6c465968fb3c9a9cb48cffbf Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 30 Apr 2026 17:25:23 -0500 Subject: [PATCH 1/3] feat: update vllm to 0.18.1 --- Dockerfile | 2 +- vllm-loadbalancer-ep.code-workspace | 11 +++++++++++ 2 files changed, 12 insertions(+), 1 deletion(-) create mode 100644 vllm-loadbalancer-ep.code-workspace diff --git a/Dockerfile b/Dockerfile index e59a03e..332bf48 100644 --- a/Dockerfile +++ b/Dockerfile @@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/ # Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.17.1" --extra-index-url https://download.pytorch.org/whl/cu129 + uv pip install --system "vllm[flashinfer]==0.18.1" --extra-index-url https://download.pytorch.org/whl/cu129 # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) COPY builder/requirements.txt /requirements.txt diff --git a/vllm-loadbalancer-ep.code-workspace b/vllm-loadbalancer-ep.code-workspace new file mode 100644 index 0000000..c29e4b2 --- /dev/null +++ b/vllm-loadbalancer-ep.code-workspace @@ -0,0 +1,11 @@ +{ + "folders": [ + { + "path": "../vllm-loadbalancer-ep" + }, + { + "path": "." + } + ], + "settings": {} +} \ No newline at end of file From a774cefe854de2cc88393967e3167adaac5742a2 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 30 Apr 2026 17:29:00 -0500 Subject: [PATCH 2/3] fix: clean workspace file --- vllm-loadbalancer-ep.code-workspace | 11 ----------- 1 file changed, 11 deletions(-) delete mode 100644 vllm-loadbalancer-ep.code-workspace diff --git a/vllm-loadbalancer-ep.code-workspace b/vllm-loadbalancer-ep.code-workspace deleted file mode 100644 index c29e4b2..0000000 --- a/vllm-loadbalancer-ep.code-workspace +++ /dev/null @@ -1,11 +0,0 @@ -{ - "folders": [ - { - "path": "../vllm-loadbalancer-ep" - }, - { - "path": "." - } - ], - "settings": {} -} \ No newline at end of file From 72547aa3bb27a9fb946435525353258556f151bb Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 30 Apr 2026 20:25:02 -0500 Subject: [PATCH 3/3] chore: update readme vllm version --- .runpod/README.md | 1 + README.md | 2 +- 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/.runpod/README.md b/.runpod/README.md index 6eb8652..c28a2ce 100644 --- a/.runpod/README.md +++ b/.runpod/README.md @@ -6,6 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API [![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) +Current vLLM version: [0.18.1](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) --- ## Endpoint Configuration diff --git a/README.md b/README.md index 0c7f0e5..9914c5f 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https: ![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png) -Current vLLM version: [0.16.0](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) +Current vLLM version: [0.18.1](https://github.com/vllm-project/vllm/releases/tag/v0.16.0) > Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)