Merge pull request #309 from adithyaJRunpod/feature/tuned-configs-round2

added configs for  Gemma 4 31B and GPT-OSS 120B
This commit is contained in:
Hailong Yang
2026-06-22 17:41:50 -04:00
committed by GitHub
2 changed files with 20 additions and 0 deletions
+11
View File
@@ -0,0 +1,11 @@
model: google/gemma-4-31b-it
gpu-memory-utilization: 0.95
max-model-len: 8192
dtype: auto
trust-remote-code: true
quantization: fp8
kv-cache-dtype: fp8
enforce-eager: false
enable-prefix-caching: true
enable-chunked-prefill: true
speculative-config: '{"model":"RedHatAI/gemma-4-31B-it-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
+9
View File
@@ -0,0 +1,9 @@
model: openai/gpt-oss-120b
gpu-memory-utilization: 0.95
max-model-len: 8192
dtype: auto
trust-remote-code: true
enforce-eager: false
enable-prefix-caching: true
enable-chunked-prefill: true
speculative-config: '{"model":"RedHatAI/gpt-oss-120b-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'