Preparing for 0.3.0
This commit is contained in:
+1
-1
@@ -1,5 +1,5 @@
|
||||
ARG WORKER_CUDA_VERSION=11.8.0
|
||||
FROM runpod/worker-vllm:base-0.2.2-cuda${WORKER_CUDA_VERSION} AS vllm-base
|
||||
FROM runpod/worker-vllm:base-0.3.0-cuda${WORKER_CUDA_VERSION} AS vllm-base
|
||||
|
||||
RUN apt-get update -y \
|
||||
&& apt-get install -y python3-pip
|
||||
|
||||
@@ -20,6 +20,13 @@ Deploy Blazing-fast LLMs powered by [vLLM](https://github.com/vllm-project/vllm)
|
||||
- [Option 1: Deploy Any Model Using Pre-Built Docker Image [Recommended]](#option-1-deploy-any-model-using-pre-built-docker-image-recommended)
|
||||
- [Prerequisites](#prerequisites)
|
||||
- [Environment Variables](#environment-variables)
|
||||
- [LLM Settings](#llm-settings)
|
||||
- [Tokenizer Settings](#tokenizer-settings)
|
||||
- [Tensor Parallelism (Multi-GPU) Settings](#tensor-parallelism-multi-gpu-settings)
|
||||
- [System Settings](#system-settings)
|
||||
- [Streaming Batch Size](#streaming-batch-size)
|
||||
- [OpenAI Settings](#openai-settings)
|
||||
- [Serverless Settings](#serverless-settings)
|
||||
- [Option 2: Build Docker Image with Model Inside](#option-2-build-docker-image-with-model-inside)
|
||||
- [Prerequisites](#prerequisites-1)
|
||||
- [Arguments](#arguments)
|
||||
@@ -76,13 +83,12 @@ Development Image: ```runpod/worker-vllm:dev```
|
||||
- `TOKENIZER_REVISION`: Tokenizer revision to load (default: `None`).
|
||||
- `CUSTOM_CHAT_TEMPLATE`: Custom chat jinja template, read more about Hugging Face chat templates [here](https://huggingface.co/docs/transformers/chat_templating). (default: `None`)
|
||||
|
||||
- Tensor Parallelism:
|
||||
- Tensor Parallelism (Multi-GPU) Settings:
|
||||
Note that the more GPUs you split a model's weights across, the slower it will be due to inter-GPU communication overhead. If you can fit the model on a single GPU, it is recommended to do so.
|
||||
- `TENSOR_PARALLEL_SIZE`: Number of GPUs to shard the model across (default: `1`).
|
||||
- If you are having issues loading your model with Tensor Parallelism, try decreasing `VLLM_CPU_FRACTION` (default: `1`).
|
||||
|
||||
- System Settings:
|
||||
- `GPU_MEMORY_UTILIZATION`: GPU VRAM utilization (default: `0.98`).
|
||||
- `GPU_MEMORY_UTILIZATION`: GPU VRAM utilization (default: `0.95`).
|
||||
- `MAX_PARALLEL_LOADING_WORKERS`: Maximum number of parallel workers for loading models, for non-Tensor Parallel only. (default: `number of available CPU cores` if `TENSOR_PARALLEL_SIZE` is `1`, otherwise `None`).
|
||||
|
||||
- Streaming Batch Size:
|
||||
|
||||
+3
-11
@@ -3,7 +3,6 @@ from dotenv import load_dotenv
|
||||
from utils import count_physical_cores
|
||||
from torch.cuda import device_count
|
||||
|
||||
|
||||
class EngineConfig:
|
||||
def __init__(self):
|
||||
load_dotenv()
|
||||
@@ -39,21 +38,14 @@ class EngineConfig:
|
||||
"gpu_memory_utilization": float(os.getenv("GPU_MEMORY_UTILIZATION", 0.95)),
|
||||
"max_parallel_loading_workers": self._get_max_parallel_loading_workers(),
|
||||
"max_model_len": self._get_max_model_len(),
|
||||
"tensor_parallel_size": self._get_num_gpu_shard(),
|
||||
"tensor_parallel_size": device_count(),
|
||||
}
|
||||
|
||||
def _get_max_parallel_loading_workers(self):
|
||||
if int(os.getenv("TENSOR_PARALLEL_SIZE", 1)) > 1:
|
||||
if device_count() > 1:
|
||||
return None
|
||||
return int(os.getenv("MAX_PARALLEL_LOADING_WORKERS", count_physical_cores()))
|
||||
|
||||
def _get_num_gpu_shard(self):
|
||||
num_gpu_shard = int(os.getenv("TENSOR_PARALLEL_SIZE", 1))
|
||||
if num_gpu_shard > 1:
|
||||
num_gpu_available = device_count()
|
||||
num_gpu_shard = min(num_gpu_shard, num_gpu_available)
|
||||
return num_gpu_shard
|
||||
|
||||
def _get_max_model_len(self):
|
||||
max_model_len = os.getenv("MAX_MODEL_LENGTH")
|
||||
return int(max_model_len) if max_model_len else None
|
||||
return int(max_model_len) if max_model_len else None
|
||||
@@ -17,6 +17,10 @@ ARG WORKER_CUDA_VERSION
|
||||
RUN apt-get update -y \
|
||||
&& apt-get install -y python3-pip git
|
||||
|
||||
RUN if [ "${WORKER_CUDA_VERSION}" = "12.1.0" ]; then \
|
||||
ldconfig /usr/local/cuda-12.1/compat/; \
|
||||
fi
|
||||
|
||||
# Set working directory
|
||||
WORKDIR /vllm-installation
|
||||
|
||||
@@ -25,6 +29,11 @@ COPY vllm-${WORKER_CUDA_VERSION}/requirements.txt requirements.txt
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
pip install -r requirements.txt
|
||||
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
if [ "${WORKER_CUDA_VERSION}" = "11.8.0" ]; then \
|
||||
pip install -U --force-reinstall torch==2.1.2 xformers==0.0.23.post1 --index-url https://download.pytorch.org/whl/cu118; \
|
||||
fi
|
||||
|
||||
# Install development dependencies
|
||||
COPY vllm-${WORKER_CUDA_VERSION}/requirements-dev.txt requirements-dev.txt
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
|
||||
Reference in New Issue
Block a user