feat: upgrade vLLM to 0.20.0

- Bump vllm[flashinfer] to 0.20.0 in Dockerfile
- Remove io_processor param from OpenAIServingRender (dropped in 0.20.0)

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
velaraptor-runpod
2026-04-30 18:39:07 -05:00
co-authored by Claude Sonnet 4.6
parent fa42ecd79a
commit 22356ee2b3
2 changed files with 1 additions and 2 deletions
+1 -1
View File
@@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/
# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels # Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels
RUN uv pip install --system "packaging>=24.2" && \ RUN uv pip install --system "packaging>=24.2" && \
uv pip install --system "vllm[flashinfer]==0.19.1" --extra-index-url https://download.pytorch.org/whl/cu129 uv pip install --system "vllm[flashinfer]==0.20.0" --extra-index-url https://download.pytorch.org/whl/cu129
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
COPY builder/requirements.txt /requirements.txt COPY builder/requirements.txt /requirements.txt
-1
View File
@@ -285,7 +285,6 @@ class OpenAIvLLMEngine(vLLMEngine):
self.openai_serving_render = OpenAIServingRender( self.openai_serving_render = OpenAIServingRender(
model_config=self.llm.model_config, model_config=self.llm.model_config,
renderer=self.llm.renderer, renderer=self.llm.renderer,
io_processor=self.llm.io_processor,
model_registry=self.serving_models.registry, model_registry=self.serving_models.registry,
request_logger=None, request_logger=None,
chat_template=chat_template, chat_template=chat_template,