fix: pin models to version, except for gpt-oss
This commit is contained in:
@@ -48,7 +48,7 @@ jobs:
|
||||
|
||||
- name: Run serverless e2e test
|
||||
id: e2e_test
|
||||
run: python scripts/serverless_e2e_test.py --config configs/qwen/qwen3_8b.yaml --build
|
||||
run: python scripts/serverless_e2e_test.py --config configs/gpt-oss/gpt_oss_120b.yaml --build
|
||||
env:
|
||||
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY_2 }}
|
||||
DOCKERHUB_REPO: ${{ vars.DOCKERHUB_REPO || 'runpod' }}
|
||||
|
||||
@@ -8,4 +8,5 @@ kv-cache-dtype: fp8
|
||||
enforce-eager: false
|
||||
enable-prefix-caching: true
|
||||
enable-chunked-prefill: true
|
||||
vllm-release: v2.22.5
|
||||
speculative-config: '{"model":"RedHatAI/gemma-4-31B-it-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||
|
||||
@@ -6,5 +6,6 @@ trust-remote-code: true
|
||||
quantization: fp8
|
||||
kv-cache-dtype: fp8
|
||||
enforce-eager: false
|
||||
vllm-release: v2.22.5
|
||||
enable-prefix-caching: true
|
||||
speculative-config: '{"model":"RedHatAI/Llama-3.1-8B-Instruct-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|
||||
|
||||
Reference in New Issue
Block a user