added configs for Gemma 4 31B and GPT-OSS 120B

This commit is contained in:
AdithyaJob
2026-06-17 22:19:58 -07:00
parent 1b3228a2dc
commit 84ec446493
2 changed files with 20 additions and 0 deletions
+11
View File
@@ -0,0 +1,11 @@
model: google/gemma-4-31b-it
gpu-memory-utilization: 0.95
max-model-len: 8192
dtype: auto
trust-remote-code: true
quantization: fp8
kv-cache-dtype: fp8
enforce-eager: false
enable-prefix-caching: true
enable-chunked-prefill: true
speculative-config: '{"model":"RedHatAI/gemma-4-31B-it-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
+9
View File
@@ -0,0 +1,9 @@
model: openai/gpt-oss-120b
gpu-memory-utilization: 0.95
max-model-len: 8192
dtype: auto
trust-remote-code: true
enforce-eager: false
enable-prefix-caching: true
enable-chunked-prefill: true
speculative-config: '{"model":"RedHatAI/gpt-oss-120b-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'