Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6d6cbe7095 | ||
|
|
90c16b472d | ||
|
|
6f2381a9a1 | ||
|
|
3851d53f93 |
@@ -929,6 +929,16 @@
|
|||||||
"advanced": true
|
"advanced": true
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
"key": "ENABLE_EXPERT_PARALLEL",
|
||||||
|
"input": {
|
||||||
|
"name": "Enable Expert Parallel",
|
||||||
|
"type": "boolean",
|
||||||
|
"description": "Enable Expert Parallel for MoE models",
|
||||||
|
"default": false,
|
||||||
|
"advanced": true
|
||||||
|
}
|
||||||
|
},
|
||||||
{
|
{
|
||||||
"key": "MODEL_REVISION",
|
"key": "MODEL_REVISION",
|
||||||
"input": {
|
"input": {
|
||||||
|
|||||||
+4
-5
@@ -1,9 +1,9 @@
|
|||||||
FROM nvidia/cuda:12.1.0-base-ubuntu22.04
|
FROM nvidia/cuda:12.4.1-base-ubuntu22.04
|
||||||
|
|
||||||
RUN apt-get update -y \
|
RUN apt-get update -y \
|
||||||
&& apt-get install -y python3-pip
|
&& apt-get install -y python3-pip
|
||||||
|
|
||||||
RUN ldconfig /usr/local/cuda-12.1/compat/
|
RUN ldconfig /usr/local/cuda-12.4/compat/
|
||||||
|
|
||||||
# Install Python dependencies
|
# Install Python dependencies
|
||||||
COPY builder/requirements.txt /requirements.txt
|
COPY builder/requirements.txt /requirements.txt
|
||||||
@@ -11,9 +11,8 @@ RUN --mount=type=cache,target=/root/.cache/pip \
|
|||||||
python3 -m pip install --upgrade pip && \
|
python3 -m pip install --upgrade pip && \
|
||||||
python3 -m pip install --upgrade -r /requirements.txt
|
python3 -m pip install --upgrade -r /requirements.txt
|
||||||
|
|
||||||
# Install vLLM (switching back to pip installs since issues that required building fork are fixed and space optimization is not as important since caching) and FlashInfer
|
# Install vLLM
|
||||||
RUN python3 -m pip install vllm==0.11.0 && \
|
RUN python3 -m pip install vllm==0.11.0
|
||||||
python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3
|
|
||||||
|
|
||||||
# Setup for Option 2: Building the Image with the Model included
|
# Setup for Option 2: Building the Image with the Model included
|
||||||
ARG MODEL_NAME=""
|
ARG MODEL_NAME=""
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
ray
|
ray
|
||||||
pandas
|
pandas
|
||||||
pyarrow
|
pyarrow
|
||||||
runpod~=1.7.7
|
runpod>=1.8,<2.0
|
||||||
huggingface-hub
|
huggingface-hub
|
||||||
packaging
|
packaging
|
||||||
typing-extensions>=4.8.0
|
typing-extensions>=4.8.0
|
||||||
|
|||||||
@@ -85,6 +85,7 @@ Complete guide to all environment variables and configuration options for worker
|
|||||||
| `ENFORCE_EAGER` | False | `bool` | Always use eager-mode PyTorch. If False(`0`), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility. |
|
| `ENFORCE_EAGER` | False | `bool` | Always use eager-mode PyTorch. If False(`0`), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility. |
|
||||||
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
|
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
|
||||||
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
|
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
|
||||||
|
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models |
|
||||||
|
|
||||||
## Tokenizer Settings
|
## Tokenizer Settings
|
||||||
|
|
||||||
|
|||||||
@@ -80,6 +80,7 @@ DEFAULT_ARGS = {
|
|||||||
"guided_decoding_backend": os.getenv('GUIDED_DECODING_BACKEND', 'outlines'),
|
"guided_decoding_backend": os.getenv('GUIDED_DECODING_BACKEND', 'outlines'),
|
||||||
"speculative_model": os.getenv('SPECULATIVE_MODEL', None),
|
"speculative_model": os.getenv('SPECULATIVE_MODEL', None),
|
||||||
"speculative_draft_tensor_parallel_size": int(os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE', 0)) or None,
|
"speculative_draft_tensor_parallel_size": int(os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE', 0)) or None,
|
||||||
|
"enable_expert_parallel": bool(os.getenv('ENABLE_EXPERT_PARALLEL', 'False').lower() == 'true'),
|
||||||
"num_speculative_tokens": int(os.getenv('NUM_SPECULATIVE_TOKENS', 0)) or None,
|
"num_speculative_tokens": int(os.getenv('NUM_SPECULATIVE_TOKENS', 0)) or None,
|
||||||
"speculative_max_model_len": int(os.getenv('SPECULATIVE_MAX_MODEL_LEN', 0)) or None,
|
"speculative_max_model_len": int(os.getenv('SPECULATIVE_MAX_MODEL_LEN', 0)) or None,
|
||||||
"speculative_disable_by_batch_size": int(os.getenv('SPECULATIVE_DISABLE_BY_BATCH_SIZE', 0)) or None,
|
"speculative_disable_by_batch_size": int(os.getenv('SPECULATIVE_DISABLE_BY_BATCH_SIZE', 0)) or None,
|
||||||
|
|||||||
Reference in New Issue
Block a user