From 08580e7ccfe6c0cccafd7f9e4304708ac3527006 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Fri, 12 Jun 2026 15:18:36 -0500 Subject: [PATCH] chore: carry non-vllm changes from main (tests GPU + configs) Brings forward the L40 GPU type in tests.json and the new llama/qwen tuned config files, while keeping Dockerfile pinned at vllm 0.20.2 (v2.20.1 state). Co-Authored-By: Claude Sonnet 4.6 --- .runpod/tests.json | 2 +- configs/llama/llama3.1_8b_instruct.yaml | 10 ++++++++++ configs/qwen/qwen3_8b.yaml | 10 ++++++++++ 3 files changed, 21 insertions(+), 1 deletion(-) create mode 100644 configs/llama/llama3.1_8b_instruct.yaml create mode 100644 configs/qwen/qwen3_8b.yaml diff --git a/.runpod/tests.json b/.runpod/tests.json index a1a5fd2..95addee 100644 --- a/.runpod/tests.json +++ b/.runpod/tests.json @@ -30,7 +30,7 @@ } ], "config": { - "gpuTypeId": "NVIDIA GeForce RTX 4090", + "gpuTypeId": "NVIDIA L40", "gpuCount": 1, "env": [ { diff --git a/configs/llama/llama3.1_8b_instruct.yaml b/configs/llama/llama3.1_8b_instruct.yaml new file mode 100644 index 0000000..76136e2 --- /dev/null +++ b/configs/llama/llama3.1_8b_instruct.yaml @@ -0,0 +1,10 @@ +model: meta-llama/Llama-3.1-8B-Instruct +gpu-memory-utilization: 0.95 +max-model-len: 8192 +dtype: auto +trust-remote-code: true +quantization: fp8 +kv-cache-dtype: fp8 +enforce-eager: false +enable-prefix-caching: true +speculative-config: '{"model":"RedHatAI/Llama-3.1-8B-Instruct-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}' diff --git a/configs/qwen/qwen3_8b.yaml b/configs/qwen/qwen3_8b.yaml new file mode 100644 index 0000000..0be2b36 --- /dev/null +++ b/configs/qwen/qwen3_8b.yaml @@ -0,0 +1,10 @@ +model: Qwen/Qwen3-8B +gpu-memory-utilization: 0.95 +max-model-len: 8192 +dtype: auto +trust-remote-code: true +quantization: fp8 +kv-cache-dtype: fp8 +enforce-eager: false +enable-prefix-caching: true +speculative-config: '{"model":"RedHatAI/Qwen3-8B-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'