diff --git a/configs/llama/llama3.1_8b_instruct.yaml b/configs/llama/llama3.1_8b_instruct.yaml new file mode 100644 index 0000000..76136e2 --- /dev/null +++ b/configs/llama/llama3.1_8b_instruct.yaml @@ -0,0 +1,10 @@ +model: meta-llama/Llama-3.1-8B-Instruct +gpu-memory-utilization: 0.95 +max-model-len: 8192 +dtype: auto +trust-remote-code: true +quantization: fp8 +kv-cache-dtype: fp8 +enforce-eager: false +enable-prefix-caching: true +speculative-config: '{"model":"RedHatAI/Llama-3.1-8B-Instruct-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}' diff --git a/configs/qwen/qwen3_8b.yaml b/configs/qwen/qwen3_8b.yaml new file mode 100644 index 0000000..0be2b36 --- /dev/null +++ b/configs/qwen/qwen3_8b.yaml @@ -0,0 +1,10 @@ +model: Qwen/Qwen3-8B +gpu-memory-utilization: 0.95 +max-model-len: 8192 +dtype: auto +trust-remote-code: true +quantization: fp8 +kv-cache-dtype: fp8 +enforce-eager: false +enable-prefix-caching: true +speculative-config: '{"model":"RedHatAI/Qwen3-8B-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'