From 88490e9551025b7725d63fd128864b46112dd822 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Thu, 9 Jul 2026 18:22:40 -0500 Subject: [PATCH] feat: update to 0.23.0 --- Dockerfile | 3 ++- configs/qwen/qwen3_8b.yaml | 2 ++ 2 files changed, 4 insertions(+), 1 deletion(-) diff --git a/Dockerfile b/Dockerfile index 9c56e20..9b99034 100644 --- a/Dockerfile +++ b/Dockerfile @@ -16,7 +16,8 @@ ENV PATH="/root/.local/bin:$PATH" RUN ldconfig /usr/local/cuda-13.0/compat/ # Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels -RUN uv pip install --system "packaging>=24.2" && \ +RUN --mount=type=cache,target=/root/.cache/uv \ + uv pip install --system "packaging>=24.2" && \ uv pip install --system "vllm[flashinfer]==0.23.0" && \ uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation && \ uv pip install --system --force-reinstall --no-deps nixl-cu13 diff --git a/configs/qwen/qwen3_8b.yaml b/configs/qwen/qwen3_8b.yaml index 0be2b36..8d6dea5 100644 --- a/configs/qwen/qwen3_8b.yaml +++ b/configs/qwen/qwen3_8b.yaml @@ -7,4 +7,6 @@ quantization: fp8 kv-cache-dtype: fp8 enforce-eager: false enable-prefix-caching: true +vllm-release: v2.22.5 +compilation-config: '{"cudagraph_mode": "PIECEWISE"}' speculative-config: '{"model":"RedHatAI/Qwen3-8B-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'