add ENABLE_EXPERT_PARALLEL engine arg for MoE models (#239)
Release / release (push) Waiting to run

* enable expert parallel arg for moe models

* add ENABLE_EXPERT_PARALLEL to hub config
This commit is contained in:
chrisvela
2025-11-17 19:25:19 +01:00
committed by GitHub
parent c896438f21
commit 3851d53f93
3 changed files with 12 additions and 0 deletions
+10
View File
@@ -929,6 +929,16 @@
"advanced": true
}
},
{
"key": "ENABLE_EXPERT_PARALLEL",
"input": {
"name": "Enable Expert Parallel",
"type": "boolean",
"description": "Enable Expert Parallel for MoE models",
"default": false,
"advanced": true
}
},
{
"key": "MODEL_REVISION",
"input": {
+1
View File
@@ -85,6 +85,7 @@ Complete guide to all environment variables and configuration options for worker
| `ENFORCE_EAGER` | False | `bool` | Always use eager-mode PyTorch. If False(`0`), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility. |
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models |
## Tokenizer Settings
+1
View File
@@ -80,6 +80,7 @@ DEFAULT_ARGS = {
"guided_decoding_backend": os.getenv('GUIDED_DECODING_BACKEND', 'outlines'),
"speculative_model": os.getenv('SPECULATIVE_MODEL', None),
"speculative_draft_tensor_parallel_size": int(os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE', 0)) or None,
"enable_expert_parallel": bool(os.getenv('ENABLE_EXPERT_PARALLEL', 'False').lower() == 'true'),
"num_speculative_tokens": int(os.getenv('NUM_SPECULATIVE_TOKENS', 0)) or None,
"speculative_max_model_len": int(os.getenv('SPECULATIVE_MAX_MODEL_LEN', 0)) or None,
"speculative_disable_by_batch_size": int(os.getenv('SPECULATIVE_DISABLE_BY_BATCH_SIZE', 0)) or None,