12 lines
363 B
YAML
12 lines
363 B
YAML
model: meta-llama/Llama-3.1-8B-Instruct
|
|
gpu-memory-utilization: 0.95
|
|
max-model-len: 8192
|
|
dtype: auto
|
|
trust-remote-code: true
|
|
quantization: fp8
|
|
kv-cache-dtype: fp8
|
|
enforce-eager: false
|
|
vllm-release: v2.22.5
|
|
enable-prefix-caching: true
|
|
speculative-config: '{"model":"RedHatAI/Llama-3.1-8B-Instruct-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
|