feat: update to 0.23.0
This commit is contained in:
+2
-1
@@ -16,7 +16,8 @@ ENV PATH="/root/.local/bin:$PATH"
|
|||||||
RUN ldconfig /usr/local/cuda-13.0/compat/
|
RUN ldconfig /usr/local/cuda-13.0/compat/
|
||||||
|
|
||||||
# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
|
# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
|
||||||
RUN uv pip install --system "packaging>=24.2" && \
|
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||||
|
uv pip install --system "packaging>=24.2" && \
|
||||||
uv pip install --system "vllm[flashinfer]==0.23.0" && \
|
uv pip install --system "vllm[flashinfer]==0.23.0" && \
|
||||||
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation && \
|
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation && \
|
||||||
uv pip install --system --force-reinstall --no-deps nixl-cu13
|
uv pip install --system --force-reinstall --no-deps nixl-cu13
|
||||||
|
|||||||
@@ -7,4 +7,6 @@ quantization: fp8
|
|||||||
kv-cache-dtype: fp8
|
kv-cache-dtype: fp8
|
||||||
enforce-eager: false
|
enforce-eager: false
|
||||||
enable-prefix-caching: true
|
enable-prefix-caching: true
|
||||||
|
vllm-release: v2.22.5
|
||||||
|
compilation-config: '{"cudagraph_mode": "PIECEWISE"}'
|
||||||
speculative-config: '{"model":"RedHatAI/Qwen3-8B-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
speculative-config: '{"model":"RedHatAI/Qwen3-8B-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||||
|
|||||||
Reference in New Issue
Block a user