feat: upgrade vLLM to 0.20.0
- Bump vllm[flashinfer] to 0.20.0 in Dockerfile - Remove io_processor param from OpenAIServingRender (dropped in 0.20.0) Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 4.6
parent
fa42ecd79a
commit
22356ee2b3
+1
-1
@@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/
|
|||||||
|
|
||||||
# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels
|
# Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels
|
||||||
RUN uv pip install --system "packaging>=24.2" && \
|
RUN uv pip install --system "packaging>=24.2" && \
|
||||||
uv pip install --system "vllm[flashinfer]==0.19.1" --extra-index-url https://download.pytorch.org/whl/cu129
|
uv pip install --system "vllm[flashinfer]==0.20.0" --extra-index-url https://download.pytorch.org/whl/cu129
|
||||||
|
|
||||||
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
|
||||||
COPY builder/requirements.txt /requirements.txt
|
COPY builder/requirements.txt /requirements.txt
|
||||||
|
|||||||
@@ -285,7 +285,6 @@ class OpenAIvLLMEngine(vLLMEngine):
|
|||||||
self.openai_serving_render = OpenAIServingRender(
|
self.openai_serving_render = OpenAIServingRender(
|
||||||
model_config=self.llm.model_config,
|
model_config=self.llm.model_config,
|
||||||
renderer=self.llm.renderer,
|
renderer=self.llm.renderer,
|
||||||
io_processor=self.llm.io_processor,
|
|
||||||
model_registry=self.serving_models.registry,
|
model_registry=self.serving_models.registry,
|
||||||
request_logger=None,
|
request_logger=None,
|
||||||
chat_template=chat_template,
|
chat_template=chat_template,
|
||||||
|
|||||||
Reference in New Issue
Block a user