diff --git a/Dockerfile b/Dockerfile index e59a03e..332bf48 100644 --- a/Dockerfile +++ b/Dockerfile @@ -10,7 +10,7 @@ RUN ldconfig /usr/local/cuda-12.9/compat/ # Install vLLM with FlashInfer - use CUDA 12.9 PyTorch wheels RUN uv pip install --system "packaging>=24.2" && \ - uv pip install --system "vllm[flashinfer]==0.17.1" --extra-index-url https://download.pytorch.org/whl/cu129 + uv pip install --system "vllm[flashinfer]==0.18.1" --extra-index-url https://download.pytorch.org/whl/cu129 # Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts) COPY builder/requirements.txt /requirements.txt diff --git a/vllm-loadbalancer-ep.code-workspace b/vllm-loadbalancer-ep.code-workspace new file mode 100644 index 0000000..c29e4b2 --- /dev/null +++ b/vllm-loadbalancer-ep.code-workspace @@ -0,0 +1,11 @@ +{ + "folders": [ + { + "path": "../vllm-loadbalancer-ep" + }, + { + "path": "." + } + ], + "settings": {} +} \ No newline at end of file