Compare commits

..
6 Commits
Author SHA1 Message Date
chrisvelaandGitHub d3a962c33b Merge pull request #301 from adithyaJRunpod/feature/tuned-configs
Release / release (push) Waiting to run
Add tuned configs,  CON-239
2026-06-11 14:26:00 -05:00
chrisvelaandGitHub 105c125698 Merge pull request #300 from runpod-workers/feat/0.21.0
feat: upgrade vllm to 0.21.0
2026-06-11 12:47:08 -05:00
AdithyaJob 0a0ccfcb60 Add tuned configs for Llama 3.1 8B and Qwen3 8B 2026-06-10 21:07:41 -07:00
velaraptor-runpod c8ce53c72c fix: add kenels, and fix for cuda 2026-06-10 16:01:53 -05:00
velaraptor-runpod 9618e799ba chore: fix cuda libraries 2026-06-04 15:35:58 -05:00
velaraptor-runpod cb3f077dba feat: upgrade vllm to 0.21.0 2026-06-03 14:43:56 -05:00
4 changed files with 33 additions and 2 deletions
+12 -1
View File
@@ -8,9 +8,20 @@ ENV PATH="/root/.local/bin:$PATH"
RUN ldconfig /usr/local/cuda-13.0/compat/
# nixl_ep PyPI wheels are compiled against CUDA 12.x and require libcudart.so.12.
# CUDA 13 runtime is ABI-compatible with CUDA 12, so symlinking is safe.
# Symlink into /usr/local/cuda/lib64 (already in LD_LIBRARY_PATH) so the linker
# finds it by filename scan rather than relying on ldcache SONAME lookup.
RUN ln -sf /usr/local/cuda/lib64/libcudart.so.13 /usr/local/cuda/lib64/libcudart.so.12 && ldconfig
# CUDA 13.0 containers return libs to /usr/local/nvidia/lib64 so container
# providers (RunPod, Lambda, etc.) can mount host drivers there consistently.
# See: https://github.com/vllm-project/vllm/issues/18859
ENV LD_LIBRARY_PATH=/usr/local/nvidia/lib64:/usr/local/cuda/lib64:$LD_LIBRARY_PATH
# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
RUN uv pip install --system "packaging>=24.2" && \
uv pip install --system "vllm[flashinfer]==0.20.2" && \
uv pip install --system "vllm[flashinfer]==0.21.0" && \
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
+1 -1
View File
@@ -3,7 +3,7 @@ pandas
pyarrow
runpod==1.9.1
huggingface-hub
lmcache==0.4.5
lmcache==0.4.6
packaging>=24.2
typing-extensions>=4.8.0
pydantic
+10
View File
@@ -0,0 +1,10 @@
model: meta-llama/Llama-3.1-8B-Instruct
gpu-memory-utilization: 0.95
max-model-len: 8192
dtype: auto
trust-remote-code: true
quantization: fp8
kv-cache-dtype: fp8
enforce-eager: false
enable-prefix-caching: true
speculative-config: '{"model":"RedHatAI/Llama-3.1-8B-Instruct-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
+10
View File
@@ -0,0 +1,10 @@
model: Qwen/Qwen3-8B
gpu-memory-utilization: 0.95
max-model-len: 8192
dtype: auto
trust-remote-code: true
quantization: fp8
kv-cache-dtype: fp8
enforce-eager: false
enable-prefix-caching: true
speculative-config: '{"model":"RedHatAI/Qwen3-8B-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'