Files
worker-vllm/configs/qwen/qwen3_8b.yaml
T

13 lines
385 B
YAML

model: Qwen/Qwen3-8B
gpu-memory-utilization: 0.95
max-model-len: 8192
dtype: auto
trust-remote-code: true
quantization: fp8
kv-cache-dtype: fp8
enforce-eager: false
enable-prefix-caching: true
vllm-release: v2.22.5
compilation-config: '{"cudagraph_mode": "PIECEWISE"}'
speculative-config: '{"model":"RedHatAI/Qwen3-8B-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'