Merge branch 'main' into fix/fix-old-actions
This commit is contained in:
+1
-1
@@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
|
||||
|
||||
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
||||
|
||||
Current vLLM version: [0.17.1](https://github.com/vllm-project/vllm/releases/tag/v0.16.0)
|
||||
Current vLLM version: [0.18.1](https://github.com/vllm-project/vllm/releases/tag/v0.16.0)
|
||||
|
||||
---
|
||||
|
||||
|
||||
+1
-1
@@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/
|
||||
|
||||
# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels
|
||||
RUN uv pip install --system "packaging>=24.2" && \
|
||||
uv pip install --system "vllm[flashinfer]==0.17.1" --extra-index-url https://download.pytorch.org/whl/cu129
|
||||
uv pip install --system "vllm[flashinfer]==0.18.1" --extra-index-url https://download.pytorch.org/whl/cu129
|
||||
|
||||
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
||||
COPY builder/requirements.txt /requirements.txt
|
||||
|
||||
@@ -8,7 +8,8 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
|
||||
|
||||

|
||||
|
||||
Current vLLM version: [0.17.1](https://github.com/vllm-project/vllm/releases/tag/v0.16.0)
|
||||
|
||||
Current vLLM version: [0.18.1](https://github.com/vllm-project/vllm/releases/tag/v0.16.0)
|
||||
|
||||
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user