fix: pin models to version, except for gpt-oss

This commit is contained in:
velaraptor-runpod
2026-07-09 19:21:24 -05:00
parent 2000acc0a3
commit 373847d9f9
3 changed files with 3 additions and 1 deletions
+1 -1
View File
@@ -48,7 +48,7 @@ jobs:
- name: Run serverless e2e test
id: e2e_test
run: python scripts/serverless_e2e_test.py --config configs/qwen/qwen3_8b.yaml --build
run: python scripts/serverless_e2e_test.py --config configs/gpt-oss/gpt_oss_120b.yaml --build
env:
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY_2 }}
DOCKERHUB_REPO: ${{ vars.DOCKERHUB_REPO || 'runpod' }}
+1
View File
@@ -8,4 +8,5 @@ kv-cache-dtype: fp8
enforce-eager: false
enable-prefix-caching: true
enable-chunked-prefill: true
vllm-release: v2.22.5
speculative-config: '{"model":"RedHatAI/gemma-4-31B-it-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
+1
View File
@@ -6,5 +6,6 @@ trust-remote-code: true
quantization: fp8
kv-cache-dtype: fp8
enforce-eager: false
vllm-release: v2.22.5
enable-prefix-caching: true
speculative-config: '{"model":"RedHatAI/Llama-3.1-8B-Instruct-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'