diff --git a/Dockerfile b/Dockerfile index 0790c31..20b257b 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,5 +1,5 @@ ARG WORKER_CUDA_VERSION=11.8.0 -FROM runpod/worker-vllm:base-0.2.2-cuda${WORKER_CUDA_VERSION} AS vllm-base +FROM runpod/worker-vllm:base-0.3.0-cuda${WORKER_CUDA_VERSION} AS vllm-base RUN apt-get update -y \ && apt-get install -y python3-pip diff --git a/README.md b/README.md index 82577d1..fba1e2e 100644 --- a/README.md +++ b/README.md @@ -20,6 +20,13 @@ Deploy Blazing-fast LLMs powered by [vLLM](https://github.com/vllm-project/vllm) - [Option 1: Deploy Any Model Using Pre-Built Docker Image [Recommended]](#option-1-deploy-any-model-using-pre-built-docker-image-recommended) - [Prerequisites](#prerequisites) - [Environment Variables](#environment-variables) + - [LLM Settings](#llm-settings) + - [Tokenizer Settings](#tokenizer-settings) + - [Tensor Parallelism (Multi-GPU) Settings](#tensor-parallelism-multi-gpu-settings) + - [System Settings](#system-settings) + - [Streaming Batch Size](#streaming-batch-size) + - [OpenAI Settings](#openai-settings) + - [Serverless Settings](#serverless-settings) - [Option 2: Build Docker Image with Model Inside](#option-2-build-docker-image-with-model-inside) - [Prerequisites](#prerequisites-1) - [Arguments](#arguments) @@ -76,13 +83,12 @@ Development Image: ```runpod/worker-vllm:dev``` - `TOKENIZER_REVISION`: Tokenizer revision to load (default: `None`). - `CUSTOM_CHAT_TEMPLATE`: Custom chat jinja template, read more about Hugging Face chat templates [here](https://huggingface.co/docs/transformers/chat_templating). (default: `None`) -- Tensor Parallelism: +- Tensor Parallelism (Multi-GPU) Settings: Note that the more GPUs you split a model's weights across, the slower it will be due to inter-GPU communication overhead. If you can fit the model on a single GPU, it is recommended to do so. - - `TENSOR_PARALLEL_SIZE`: Number of GPUs to shard the model across (default: `1`). - If you are having issues loading your model with Tensor Parallelism, try decreasing `VLLM_CPU_FRACTION` (default: `1`). - System Settings: - - `GPU_MEMORY_UTILIZATION`: GPU VRAM utilization (default: `0.98`). + - `GPU_MEMORY_UTILIZATION`: GPU VRAM utilization (default: `0.95`). - `MAX_PARALLEL_LOADING_WORKERS`: Maximum number of parallel workers for loading models, for non-Tensor Parallel only. (default: `number of available CPU cores` if `TENSOR_PARALLEL_SIZE` is `1`, otherwise `None`). - Streaming Batch Size: diff --git a/src/config.py b/src/config.py index 3176248..fffd730 100644 --- a/src/config.py +++ b/src/config.py @@ -3,7 +3,6 @@ from dotenv import load_dotenv from utils import count_physical_cores from torch.cuda import device_count - class EngineConfig: def __init__(self): load_dotenv() @@ -39,21 +38,14 @@ class EngineConfig: "gpu_memory_utilization": float(os.getenv("GPU_MEMORY_UTILIZATION", 0.95)), "max_parallel_loading_workers": self._get_max_parallel_loading_workers(), "max_model_len": self._get_max_model_len(), - "tensor_parallel_size": self._get_num_gpu_shard(), + "tensor_parallel_size": device_count(), } def _get_max_parallel_loading_workers(self): - if int(os.getenv("TENSOR_PARALLEL_SIZE", 1)) > 1: + if device_count() > 1: return None return int(os.getenv("MAX_PARALLEL_LOADING_WORKERS", count_physical_cores())) - def _get_num_gpu_shard(self): - num_gpu_shard = int(os.getenv("TENSOR_PARALLEL_SIZE", 1)) - if num_gpu_shard > 1: - num_gpu_available = device_count() - num_gpu_shard = min(num_gpu_shard, num_gpu_available) - return num_gpu_shard - def _get_max_model_len(self): max_model_len = os.getenv("MAX_MODEL_LENGTH") - return int(max_model_len) if max_model_len else None + return int(max_model_len) if max_model_len else None \ No newline at end of file diff --git a/vllm-base/Dockerfile b/vllm-base/Dockerfile index b6858f1..05d126f 100644 --- a/vllm-base/Dockerfile +++ b/vllm-base/Dockerfile @@ -17,6 +17,10 @@ ARG WORKER_CUDA_VERSION RUN apt-get update -y \ && apt-get install -y python3-pip git +RUN if [ "${WORKER_CUDA_VERSION}" = "12.1.0" ]; then \ + ldconfig /usr/local/cuda-12.1/compat/; \ + fi + # Set working directory WORKDIR /vllm-installation @@ -25,6 +29,11 @@ COPY vllm-${WORKER_CUDA_VERSION}/requirements.txt requirements.txt RUN --mount=type=cache,target=/root/.cache/pip \ pip install -r requirements.txt +RUN --mount=type=cache,target=/root/.cache/pip \ + if [ "${WORKER_CUDA_VERSION}" = "11.8.0" ]; then \ + pip install -U --force-reinstall torch==2.1.2 xformers==0.0.23.post1 --index-url https://download.pytorch.org/whl/cu118; \ + fi + # Install development dependencies COPY vllm-${WORKER_CUDA_VERSION}/requirements-dev.txt requirements-dev.txt RUN --mount=type=cache,target=/root/.cache/pip \