Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
90c16b472d | ||
|
|
6f2381a9a1 | ||
|
|
3851d53f93 |
@@ -929,6 +929,16 @@
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "ENABLE_EXPERT_PARALLEL",
|
||||
"input": {
|
||||
"name": "Enable Expert Parallel",
|
||||
"type": "boolean",
|
||||
"description": "Enable Expert Parallel for MoE models",
|
||||
"default": false,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "MODEL_REVISION",
|
||||
"input": {
|
||||
|
||||
+4
-5
@@ -1,9 +1,9 @@
|
||||
FROM nvidia/cuda:12.1.0-base-ubuntu22.04
|
||||
FROM nvidia/cuda:12.4.1-base-ubuntu22.04
|
||||
|
||||
RUN apt-get update -y \
|
||||
&& apt-get install -y python3-pip
|
||||
|
||||
RUN ldconfig /usr/local/cuda-12.1/compat/
|
||||
RUN ldconfig /usr/local/cuda-12.4/compat/
|
||||
|
||||
# Install Python dependencies
|
||||
COPY builder/requirements.txt /requirements.txt
|
||||
@@ -11,9 +11,8 @@ RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
python3 -m pip install --upgrade pip && \
|
||||
python3 -m pip install --upgrade -r /requirements.txt
|
||||
|
||||
# Install vLLM (switching back to pip installs since issues that required building fork are fixed and space optimization is not as important since caching) and FlashInfer
|
||||
RUN python3 -m pip install vllm==0.11.0 && \
|
||||
python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3
|
||||
# Install vLLM
|
||||
RUN python3 -m pip install vllm==0.11.0
|
||||
|
||||
# Setup for Option 2: Building the Image with the Model included
|
||||
ARG MODEL_NAME=""
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
ray
|
||||
pandas
|
||||
pyarrow
|
||||
runpod~=1.7.7
|
||||
runpod>=1.8,<2.0
|
||||
huggingface-hub
|
||||
packaging
|
||||
typing-extensions>=4.8.0
|
||||
|
||||
@@ -85,6 +85,7 @@ Complete guide to all environment variables and configuration options for worker
|
||||
| `ENFORCE_EAGER` | False | `bool` | Always use eager-mode PyTorch. If False(`0`), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility. |
|
||||
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` | Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode. |
|
||||
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` | Enables or disables custom all reduce. |
|
||||
| `ENABLE_EXPERT_PARALLEL` | `False` | `bool` | Enable Expert Parallel for MoE models |
|
||||
|
||||
## Tokenizer Settings
|
||||
|
||||
|
||||
@@ -80,6 +80,7 @@ DEFAULT_ARGS = {
|
||||
"guided_decoding_backend": os.getenv('GUIDED_DECODING_BACKEND', 'outlines'),
|
||||
"speculative_model": os.getenv('SPECULATIVE_MODEL', None),
|
||||
"speculative_draft_tensor_parallel_size": int(os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE', 0)) or None,
|
||||
"enable_expert_parallel": bool(os.getenv('ENABLE_EXPERT_PARALLEL', 'False').lower() == 'true'),
|
||||
"num_speculative_tokens": int(os.getenv('NUM_SPECULATIVE_TOKENS', 0)) or None,
|
||||
"speculative_max_model_len": int(os.getenv('SPECULATIVE_MAX_MODEL_LEN', 0)) or None,
|
||||
"speculative_disable_by_batch_size": int(os.getenv('SPECULATIVE_DISABLE_BY_BATCH_SIZE', 0)) or None,
|
||||
|
||||
Reference in New Issue
Block a user