chore: carry non-vllm changes from main (tests GPU + configs)

Brings forward the L40 GPU type in tests.json and the new llama/qwen
tuned config files, while keeping Dockerfile pinned at vllm 0.20.2
(v2.20.1 state).

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
velaraptor-runpod
2026-06-12 15:18:36 -05:00
co-authored by Claude Sonnet 4.6
parent 69646b9e99
commit 08580e7ccf
3 changed files with 21 additions and 1 deletions
+1 -1
View File
@@ -30,7 +30,7 @@
}
],
"config": {
"gpuTypeId": "NVIDIA GeForce RTX 4090",
"gpuTypeId": "NVIDIA L40",
"gpuCount": 1,
"env": [
{
+10
View File
@@ -0,0 +1,10 @@
model: meta-llama/Llama-3.1-8B-Instruct
gpu-memory-utilization: 0.95
max-model-len: 8192
dtype: auto
trust-remote-code: true
quantization: fp8
kv-cache-dtype: fp8
enforce-eager: false
enable-prefix-caching: true
speculative-config: '{"model":"RedHatAI/Llama-3.1-8B-Instruct-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
+10
View File
@@ -0,0 +1,10 @@
model: Qwen/Qwen3-8B
gpu-memory-utilization: 0.95
max-model-len: 8192
dtype: auto
trust-remote-code: true
quantization: fp8
kv-cache-dtype: fp8
enforce-eager: false
enable-prefix-caching: true
speculative-config: '{"model":"RedHatAI/Qwen3-8B-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'