Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0e1e38326a | ||
|
|
c8458fef2b | ||
|
|
bad5ddd892 | ||
|
|
f19ce12ab0 | ||
|
|
1bb6f84541 | ||
|
|
30cb56a3df | ||
|
|
00add8707a | ||
|
|
ec7ea0b760 | ||
|
|
9f2cb7b1d0 | ||
|
|
4abe494635 | ||
|
|
4f61b04afe | ||
|
|
874379a0c5 | ||
|
|
f06a64d5b9 | ||
|
|
0a5b5bc095 | ||
|
|
2936e4d95d |
@@ -1,45 +0,0 @@
|
|||||||
name: CD | Docker-Build-Release
|
|
||||||
|
|
||||||
on:
|
|
||||||
push:
|
|
||||||
branches:
|
|
||||||
- "main"
|
|
||||||
release:
|
|
||||||
types: [published]
|
|
||||||
workflow_dispatch:
|
|
||||||
inputs:
|
|
||||||
image_tag:
|
|
||||||
description: "Docker Image Tag"
|
|
||||||
required: false
|
|
||||||
default: "dev"
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
docker-build:
|
|
||||||
runs-on: DO
|
|
||||||
# DO is a custom runner deployed on DigitalOcean, only available for workflows under the runpod-workers organization.
|
|
||||||
# If you would like to use this workflow, you can replace DO with ubuntu-latest or any other runner.
|
|
||||||
|
|
||||||
strategy:
|
|
||||||
matrix:
|
|
||||||
cuda_version: [11.8.0, 12.1.0]
|
|
||||||
|
|
||||||
steps:
|
|
||||||
- name: Set up QEMU
|
|
||||||
uses: docker/setup-qemu-action@v2
|
|
||||||
|
|
||||||
- name: Set up Docker Buildx
|
|
||||||
uses: docker/setup-buildx-action@v2
|
|
||||||
|
|
||||||
- name: Login to Docker Hub
|
|
||||||
uses: docker/login-action@v2
|
|
||||||
with:
|
|
||||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
|
||||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
|
||||||
|
|
||||||
# Build and push step
|
|
||||||
- name: Build and push
|
|
||||||
uses: docker/build-push-action@v4
|
|
||||||
with:
|
|
||||||
push: true
|
|
||||||
tags: ${{ vars.DOCKERHUB_REPO }}/${{ vars.DOCKERHUB_IMG }}:${{ (github.event_name == 'release' && github.event.release.tag_name) || (github.event_name == 'workflow_dispatch' && github.event.inputs.image_tag) || 'dev' }}-cuda${{ matrix.cuda_version }}
|
|
||||||
build-args: WORKER_CUDA_VERSION=${{ matrix.cuda_version }}
|
|
||||||
+8
-6
@@ -1,5 +1,6 @@
|
|||||||
ARG WORKER_CUDA_VERSION=11.8.0
|
ARG WORKER_CUDA_VERSION=11.8.0
|
||||||
FROM runpod/worker-vllm:base-0.3.2-cuda${WORKER_CUDA_VERSION} AS vllm-base
|
ARG BASE_IMAGE_VERSION=1.0.0
|
||||||
|
FROM runpod/worker-vllm:base-${BASE_IMAGE_VERSION}-cuda${WORKER_CUDA_VERSION} AS vllm-base
|
||||||
|
|
||||||
RUN apt-get update -y \
|
RUN apt-get update -y \
|
||||||
&& apt-get install -y python3-pip
|
&& apt-get install -y python3-pip
|
||||||
@@ -19,7 +20,7 @@ ARG MODEL_REVISION=""
|
|||||||
ARG TOKENIZER_REVISION=""
|
ARG TOKENIZER_REVISION=""
|
||||||
|
|
||||||
ENV MODEL_NAME=$MODEL_NAME \
|
ENV MODEL_NAME=$MODEL_NAME \
|
||||||
MODEL_REVISION=$REVISION \
|
MODEL_REVISION=$MODEL_REVISION \
|
||||||
TOKENIZER_NAME=$TOKENIZER_NAME \
|
TOKENIZER_NAME=$TOKENIZER_NAME \
|
||||||
TOKENIZER_REVISION=$TOKENIZER_REVISION \
|
TOKENIZER_REVISION=$TOKENIZER_REVISION \
|
||||||
BASE_PATH=$BASE_PATH \
|
BASE_PATH=$BASE_PATH \
|
||||||
@@ -27,11 +28,11 @@ ENV MODEL_NAME=$MODEL_NAME \
|
|||||||
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
|
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
|
||||||
HUGGINGFACE_HUB_CACHE="${BASE_PATH}/huggingface-cache/hub" \
|
HUGGINGFACE_HUB_CACHE="${BASE_PATH}/huggingface-cache/hub" \
|
||||||
HF_HOME="${BASE_PATH}/huggingface-cache/hub" \
|
HF_HOME="${BASE_PATH}/huggingface-cache/hub" \
|
||||||
HF_TRANSFER=1
|
HF_HUB_ENABLE_HF_TRANSFER=1
|
||||||
|
|
||||||
ENV PYTHONPATH="/:/vllm-installation"
|
ENV PYTHONPATH="/:/vllm-workspace"
|
||||||
|
|
||||||
COPY builder/download_model.py /download_model.py
|
COPY src/download_model.py /download_model.py
|
||||||
RUN --mount=type=secret,id=HF_TOKEN,required=false \
|
RUN --mount=type=secret,id=HF_TOKEN,required=false \
|
||||||
if [ -f /run/secrets/HF_TOKEN ]; then \
|
if [ -f /run/secrets/HF_TOKEN ]; then \
|
||||||
export HF_TOKEN=$(cat /run/secrets/HF_TOKEN); \
|
export HF_TOKEN=$(cat /run/secrets/HF_TOKEN); \
|
||||||
@@ -42,7 +43,8 @@ RUN --mount=type=secret,id=HF_TOKEN,required=false \
|
|||||||
|
|
||||||
# Add source files
|
# Add source files
|
||||||
COPY src /src
|
COPY src /src
|
||||||
|
# Remove download_model.py
|
||||||
|
RUN rm /download_model.py
|
||||||
|
|
||||||
# Start the handler
|
# Start the handler
|
||||||
CMD ["python3", "/src/handler.py"]
|
CMD ["python3", "/src/handler.py"]
|
||||||
@@ -1,29 +1,34 @@
|
|||||||
<div align="center">
|
<div align="center">
|
||||||
|
|
||||||
<h1> vLLM Serverless Endpoint Worker </h1>
|
# OpenAI-Compatible vLLM Serverless Endpoint Worker
|
||||||
|
Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https://github.com/vllm-project/vllm) Inference Engine on RunPod Serverless with just a few clicks.
|
||||||
Deploy Blazing-fast LLMs powered by [vLLM](https://github.com/vllm-project/vllm) on RunPod Serverless in a few clicks.
|
<!--
|
||||||
|

|
||||||
<p>Worker Version: 0.3.2 | vLLM Version: 0.3.3</p>
|

|
||||||
|
\
|
||||||
[](https://github.com/runpod-workers/worker-vllm/actions/workflows/docker-build-release.yml)
|
 -->
|
||||||
|
<!--
|
||||||
|
 -->
|
||||||
|
|
||||||
|
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
### Worker vLLM 0.3.0: What's New since 0.2.0:
|
# News:
|
||||||
- **🚀 Full OpenAI Compatibility 🚀**
|
|
||||||
|
### 1. UI for Deploying vLLM Worker on RunPod console:
|
||||||
|

|
||||||
|
|
||||||
|
### 2. Worker vLLM `1.0.0` with vLLM `0.4.2` now available under `stable` tags
|
||||||
|
Update 1.0.0 is now available, use the image tag `runpod/worker-vllm:stable-cuda12.1.0` or `runpod/worker-vllm:stable-cuda11.8.0`.
|
||||||
|
|
||||||
|
### 3. OpenAI-Compatible [Embedding Worker](https://github.com/runpod-workers/worker-infinity-embedding) Released
|
||||||
|
Deploy your own OpenAI-compatible Serverless Endpoint on RunPod with multiple embedding models and fast inference for RAG and more!
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
### 4. Caching Accross RunPod Machines
|
||||||
|
Worker vLLM is now cached on all RunPod machines, resulting in near-instant deployment! Previously, downloading and extracting the image took 3-5 minutes on average.
|
||||||
|
|
||||||
You may now use your deployment with any OpenAI Codebase by changing **only 3 lines** in total. The supported routes are <ins>Chat Completions</ins>, <ins>Completions</ins>, and <ins>Models</ins> - with both streaming and non-streaming.
|
|
||||||
- **Dynamic Batch Size** - time-to-first token as fast no batching, while maintaining the performance of batched token streaming throughout the request.
|
|
||||||
- vLLM 0.2.7 -> 0.3.2
|
|
||||||
- Gemma, DeepSeek MoE and OLMo support.
|
|
||||||
- FP8 KV Cache support
|
|
||||||
- New supported parameters
|
|
||||||
- We're working on adding support for Multi-LoRA ⚙️
|
|
||||||
- Support for a wide range of new settings for your endpoint, such as Custom chat templates.
|
|
||||||
- Fixed Tensor Parallelism, baking model into images, and more bugs.
|
|
||||||
- Refactors and general improvements.
|
|
||||||
|
|
||||||
## Table of Contents
|
## Table of Contents
|
||||||
- [Setting up the Serverless Worker](#setting-up-the-serverless-worker)
|
- [Setting up the Serverless Worker](#setting-up-the-serverless-worker)
|
||||||
@@ -57,10 +62,11 @@ Deploy Blazing-fast LLMs powered by [vLLM](https://github.com/vllm-project/vllm)
|
|||||||
# Setting up the Serverless Worker
|
# Setting up the Serverless Worker
|
||||||
|
|
||||||
### Option 1: Deploy Any Model Using Pre-Built Docker Image [Recommended]
|
### Option 1: Deploy Any Model Using Pre-Built Docker Image [Recommended]
|
||||||
> [!TIP]
|
|
||||||
> This is the quickest and easiest way to tes your model, as it does not require you to build a Docker image, upload heavy models to DockerHub and wait for workers to download them. You can use this option to deploy your model in a few clicks. For even more convenience, attach a network storage volume to your Endpoint, which will download the model once and share it across all workers.
|
> [!NOTE]
|
||||||
>
|
> You can now deploy from the dedicated UI on the RunPod console with all of the settings and choices listed.
|
||||||
> However, for actual deployment, it is recommended that you build an image with the model baked in, which is described in Option 2 - this will ensure the fastest load speeds.
|
> Try now by accessing in Explore or Serverless pages on the RunPod console!
|
||||||
|
|
||||||
|
|
||||||
We now offer a pre-built Docker Image for the vLLM Worker that you can configure entirely with Environment Variables when creating the RunPod Serverless Endpoint:
|
We now offer a pre-built Docker Image for the vLLM Worker that you can configure entirely with Environment Variables when creating the RunPod Serverless Endpoint:
|
||||||
|
|
||||||
@@ -72,17 +78,17 @@ Below is a summary of the available RunPod Worker images, categorized by image s
|
|||||||
|
|
||||||
| CUDA Version | Stable Image Tag | Development Image Tag | Note |
|
| CUDA Version | Stable Image Tag | Development Image Tag | Note |
|
||||||
|--------------|-----------------------------------|-----------------------------------|----------------------------------------------------------------------|
|
|--------------|-----------------------------------|-----------------------------------|----------------------------------------------------------------------|
|
||||||
| 11.8.0 | `runpod/worker-vllm:0.3.2-cuda11.8.0` | `runpod/worker-vllm:dev-cuda11.8.0` | Available on all RunPod Workers without additional selection needed. |
|
| 11.8.0 | `runpod/worker-vllm:stable-cuda11.8.0` | `runpod/worker-vllm:dev-cuda11.8.0` | Available on all RunPod Workers without additional selection needed. |
|
||||||
| 12.1.0 | `runpod/worker-vllm:0.3.2-cuda12.1.0` | `runpod/worker-vllm:dev-cuda12.1.0` | When creating an Endpoint, select CUDA Version 12.2 and 12.1 in the filter. |
|
| 12.1.0 | `runpod/worker-vllm:stable-cuda12.1.0` | `runpod/worker-vllm:dev-cuda12.1.0` | When creating an Endpoint, select CUDA Version 12.3, 12.2 and 12.1 in the filter. |
|
||||||
|
|
||||||
|
|
||||||
This table provides a quick reference to the image tags you should use based on the desired CUDA version and image stability (Stable or Development). Ensure to follow the selection note for CUDA 12.1.0 compatibility.
|
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
#### Prerequisites
|
#### Prerequisites
|
||||||
- RunPod Account
|
- RunPod Account
|
||||||
|
|
||||||
#### Environment Variables
|
#### Environment Variables/Settings
|
||||||
> Note: `0` is equivalent to `False` and `1` is equivalent to `True` for boolean values.
|
> Note: `0` is equivalent to `False` and `1` is equivalent to `True` for boolean values.
|
||||||
|
|
||||||
| Name | Default | Type/Choices | Description |
|
| Name | Default | Type/Choices | Description |
|
||||||
@@ -97,7 +103,7 @@ This table provides a quick reference to the image tags you should use based on
|
|||||||
| `QUANTIZATION` | `None` | `awq`, `squeezellm`, `gptq` |Quantization of given model. The model must already be quantized. |
|
| `QUANTIZATION` | `None` | `awq`, `squeezellm`, `gptq` |Quantization of given model. The model must already be quantized. |
|
||||||
| `TRUST_REMOTE_CODE` | `0` | boolean as `int` |Trust remote code for Hugging Face models. Can help with Mixtral 8x7B, Quantized models, and unusual models/architectures.
|
| `TRUST_REMOTE_CODE` | `0` | boolean as `int` |Trust remote code for Hugging Face models. Can help with Mixtral 8x7B, Quantized models, and unusual models/architectures.
|
||||||
| `SEED` | `0` | `int` |Sets random seed for operations. |
|
| `SEED` | `0` | `int` |Sets random seed for operations. |
|
||||||
| `KV_CACHE_DTYPE` | `auto` | boolean as `int` |Data type for kv cache storage. Uses `DTYPE` if set to `auto`. |
|
| `KV_CACHE_DTYPE` | `auto` | `auto`, `fp8` |Data type for kv cache storage. Uses `DTYPE` if set to `auto`. |
|
||||||
| `DTYPE` | `auto` | `auto`, `half`, `float16`, `bfloat16`, `float`, `float32` |Sets datatype/precision for model weights and activations. |
|
| `DTYPE` | `auto` | `auto`, `half`, `float16`, `bfloat16`, `float`, `float32` |Sets datatype/precision for model weights and activations. |
|
||||||
**Tokenizer Settings**
|
**Tokenizer Settings**
|
||||||
| `TOKENIZER_NAME` | `None` | `str` |Tokenizer repository to use a different tokenizer than the model's default. |
|
| `TOKENIZER_NAME` | `None` | `str` |Tokenizer repository to use a different tokenizer than the model's default. |
|
||||||
@@ -109,7 +115,7 @@ This table provides a quick reference to the image tags you should use based on
|
|||||||
| `BLOCK_SIZE` | `16` | `8`, `16`, `32` |Token block size for contiguous chunks of tokens. |
|
| `BLOCK_SIZE` | `16` | `8`, `16`, `32` |Token block size for contiguous chunks of tokens. |
|
||||||
| `SWAP_SPACE` | `4` | `int` |CPU swap space size (GiB) per GPU. |
|
| `SWAP_SPACE` | `4` | `int` |CPU swap space size (GiB) per GPU. |
|
||||||
| `ENFORCE_EAGER` | `0` | boolean as `int` |Always use eager-mode PyTorch. If False(`0`), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility. |
|
| `ENFORCE_EAGER` | `0` | boolean as `int` |Always use eager-mode PyTorch. If False(`0`), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility. |
|
||||||
| `MAX_CONTEXT_LEN_TO_CAPTURE` | `8192` | `int` |Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode.|
|
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` |Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode.|
|
||||||
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` |Enables or disables custom all reduce. |
|
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` |Enables or disables custom all reduce. |
|
||||||
**Streaming Batch Size Settings**:
|
**Streaming Batch Size Settings**:
|
||||||
| `DEFAULT_BATCH_SIZE` | `50` | `int` |Default and Maximum batch size for token streaming to reduce HTTP calls. |
|
| `DEFAULT_BATCH_SIZE` | `50` | `int` |Default and Maximum batch size for token streaming to reduce HTTP calls. |
|
||||||
@@ -176,6 +182,8 @@ Below are all supported model architectures (and examples of each) that you can
|
|||||||
- Baichuan & Baichuan2 (`baichuan-inc/Baichuan2-13B-Chat`, `baichuan-inc/Baichuan-7B`, etc.)
|
- Baichuan & Baichuan2 (`baichuan-inc/Baichuan2-13B-Chat`, `baichuan-inc/Baichuan-7B`, etc.)
|
||||||
- BLOOM (`bigscience/bloom`, `bigscience/bloomz`, etc.)
|
- BLOOM (`bigscience/bloom`, `bigscience/bloomz`, etc.)
|
||||||
- ChatGLM (`THUDM/chatglm2-6b`, `THUDM/chatglm3-6b`, etc.)
|
- ChatGLM (`THUDM/chatglm2-6b`, `THUDM/chatglm3-6b`, etc.)
|
||||||
|
- Command-R (`CohereForAI/c4ai-command-r-v01`, etc.)
|
||||||
|
- DBRX (`databricks/dbrx-base`, `databricks/dbrx-instruct` etc.)
|
||||||
- DeciLM (`Deci/DeciLM-7B`, `Deci/DeciLM-7B-instruct`, etc.)
|
- DeciLM (`Deci/DeciLM-7B`, `Deci/DeciLM-7B-instruct`, etc.)
|
||||||
- Falcon (`tiiuae/falcon-7b`, `tiiuae/falcon-40b`, `tiiuae/falcon-rw-7b`, etc.)
|
- Falcon (`tiiuae/falcon-7b`, `tiiuae/falcon-40b`, `tiiuae/falcon-rw-7b`, etc.)
|
||||||
- Gemma (`google/gemma-2b`, `google/gemma-7b`, etc.)
|
- Gemma (`google/gemma-2b`, `google/gemma-7b`, etc.)
|
||||||
@@ -185,16 +193,23 @@ Below are all supported model architectures (and examples of each) that you can
|
|||||||
- GPT-NeoX (`EleutherAI/gpt-neox-20b`, `databricks/dolly-v2-12b`, `stabilityai/stablelm-tuned-alpha-7b`, etc.)
|
- GPT-NeoX (`EleutherAI/gpt-neox-20b`, `databricks/dolly-v2-12b`, `stabilityai/stablelm-tuned-alpha-7b`, etc.)
|
||||||
- InternLM (`internlm/internlm-7b`, `internlm/internlm-chat-7b`, etc.)
|
- InternLM (`internlm/internlm-7b`, `internlm/internlm-chat-7b`, etc.)
|
||||||
- InternLM2 (`internlm/internlm2-7b`, `internlm/internlm2-chat-7b`, etc.)
|
- InternLM2 (`internlm/internlm2-7b`, `internlm/internlm2-chat-7b`, etc.)
|
||||||
- LLaMA & LLaMA-2 (`meta-llama/Llama-2-70b-hf`, `lmsys/vicuna-13b-v1.3`, `young-geng/koala`, `openlm-research/open_llama_13b`, etc.)
|
- Jais (`core42/jais-13b`, `core42/jais-13b-chat`, `core42/jais-30b-v3`, `core42/jais-30b-chat-v3`, etc.)
|
||||||
|
- LLaMA, Llama 2, and Meta Llama 3 (`meta-llama/Meta-Llama-3-8B-Instruct`, `meta-llama/Meta-Llama-3-70B-Instruct`, `meta-llama/Llama-2-70b-hf`, `lmsys/vicuna-13b-v1.3`, `young-geng/koala`, `openlm-research/open_llama_13b`, etc.)
|
||||||
|
- MiniCPM (`openbmb/MiniCPM-2B-sft-bf16`, `openbmb/MiniCPM-2B-dpo-bf16`, etc.)
|
||||||
- Mistral (`mistralai/Mistral-7B-v0.1`, `mistralai/Mistral-7B-Instruct-v0.1`, etc.)
|
- Mistral (`mistralai/Mistral-7B-v0.1`, `mistralai/Mistral-7B-Instruct-v0.1`, etc.)
|
||||||
- Mixtral (`mistralai/Mixtral-8x7B-v0.1`, `mistralai/Mixtral-8x7B-Instruct-v0.1`, etc.)
|
- Mixtral (`mistralai/Mixtral-8x7B-v0.1`, `mistralai/Mixtral-8x7B-Instruct-v0.1`, `mistral-community/Mixtral-8x22B-v0.1`, etc.)
|
||||||
- MPT (`mosaicml/mpt-7b`, `mosaicml/mpt-30b`, etc.)
|
- MPT (`mosaicml/mpt-7b`, `mosaicml/mpt-30b`, etc.)
|
||||||
- OLMo (`allenai/OLMo-1B`, `allenai/OLMo-7B`, etc.)
|
- OLMo (`allenai/OLMo-1B-hf`, `allenai/OLMo-7B-hf`, etc.)
|
||||||
- OPT (`facebook/opt-66b`, `facebook/opt-iml-max-30b`, etc.)
|
- OPT (`facebook/opt-66b`, `facebook/opt-iml-max-30b`, etc.)
|
||||||
|
- Orion (`OrionStarAI/Orion-14B-Base`, `OrionStarAI/Orion-14B-Chat`, etc.)
|
||||||
- Phi (`microsoft/phi-1_5`, `microsoft/phi-2`, etc.)
|
- Phi (`microsoft/phi-1_5`, `microsoft/phi-2`, etc.)
|
||||||
|
- Phi-3 (`microsoft/Phi-3-mini-4k-instruct`, `microsoft/Phi-3-mini-128k-instruct`, etc.)
|
||||||
- Qwen (`Qwen/Qwen-7B`, `Qwen/Qwen-7B-Chat`, etc.)
|
- Qwen (`Qwen/Qwen-7B`, `Qwen/Qwen-7B-Chat`, etc.)
|
||||||
- Qwen2 (`Qwen/Qwen2-7B-beta`, `Qwen/Qwen-7B-Chat-beta`, etc.)
|
- Qwen2 (`Qwen/Qwen1.5-7B`, `Qwen/Qwen1.5-7B-Chat`, etc.)
|
||||||
|
- Qwen2MoE (`Qwen/Qwen1.5-MoE-A2.7B`, `Qwen/Qwen1.5-MoE-A2.7B-Chat`, etc.)
|
||||||
- StableLM(`stabilityai/stablelm-3b-4e1t`, `stabilityai/stablelm-base-alpha-7b-v2`, etc.)
|
- StableLM(`stabilityai/stablelm-3b-4e1t`, `stabilityai/stablelm-base-alpha-7b-v2`, etc.)
|
||||||
|
- Starcoder2(`bigcode/starcoder2-3b`, `bigcode/starcoder2-7b`, `bigcode/starcoder2-15b`, etc.)
|
||||||
|
- Xverse (`xverse/XVERSE-7B-Chat`, `xverse/XVERSE-13B-Chat`, `xverse/XVERSE-65B-Chat`, etc.)
|
||||||
- Yi (`01-ai/Yi-6B`, `01-ai/Yi-34B`, etc.)
|
- Yi (`01-ai/Yi-6B`, `01-ai/Yi-34B`, etc.)
|
||||||
|
|
||||||
# Usage: OpenAI Compatibility
|
# Usage: OpenAI Compatibility
|
||||||
|
|||||||
@@ -1,50 +0,0 @@
|
|||||||
import os
|
|
||||||
import shutil
|
|
||||||
from huggingface_hub import snapshot_download
|
|
||||||
from vllm.model_executor.weight_utils import prepare_hf_model_weights, Disabledtqdm
|
|
||||||
|
|
||||||
def download_extras_or_tokenizer(model_name, cache_dir, revision, extras=False):
|
|
||||||
"""Download model or tokenizer and prepare its weights, returning the local folder path."""
|
|
||||||
pattern = ["*token*", "*.json"] if extras else None
|
|
||||||
extra_dir = "/extras" if extras else ""
|
|
||||||
folder = snapshot_download(
|
|
||||||
model_name,
|
|
||||||
cache_dir=cache_dir + extra_dir,
|
|
||||||
revision=revision,
|
|
||||||
tqdm_class=Disabledtqdm,
|
|
||||||
allow_patterns=pattern if extras else None,
|
|
||||||
ignore_patterns=["*.safetensors", "*.bin", "*.pt"] if not extras else None
|
|
||||||
)
|
|
||||||
return folder
|
|
||||||
|
|
||||||
def move_files(src_dir, dest_dir):
|
|
||||||
"""Move files from source to destination directory."""
|
|
||||||
for f in os.listdir(src_dir):
|
|
||||||
src_path = os.path.join(src_dir, f)
|
|
||||||
dst_path = os.path.join(dest_dir, f)
|
|
||||||
shutil.copy2(src_path, dst_path)
|
|
||||||
os.remove(src_path)
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
model, download_dir = os.getenv("MODEL_NAME"), os.getenv("HF_HOME")
|
|
||||||
tokenizer = os.getenv("TOKENIZER_NAME") or model
|
|
||||||
|
|
||||||
revisions = {
|
|
||||||
"model": os.getenv("MODEL_REVISION") or None,
|
|
||||||
"tokenizer": os.getenv("TOKENIZER_REVISION") or None
|
|
||||||
}
|
|
||||||
|
|
||||||
if not model or not download_dir:
|
|
||||||
raise ValueError(f"Must specify model and download_dir. Model: {model}, download_dir: {download_dir}")
|
|
||||||
|
|
||||||
os.makedirs(download_dir, exist_ok=True)
|
|
||||||
model_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights(model_name_or_path=model, revision=revisions["model"], cache_dir=download_dir)
|
|
||||||
model_extras_folder = download_extras_or_tokenizer(model, download_dir, revisions["model"], extras=True)
|
|
||||||
move_files(model_extras_folder, model_folder)
|
|
||||||
|
|
||||||
with open("/local_model_path.txt", "w") as f:
|
|
||||||
f.write(model_folder)
|
|
||||||
|
|
||||||
tokenizer_folder = download_extras_or_tokenizer(tokenizer, download_dir, revisions["tokenizer"])
|
|
||||||
with open("/local_tokenizer_path.txt", "w") as f:
|
|
||||||
f.write(tokenizer_folder)
|
|
||||||
@@ -1,4 +1,3 @@
|
|||||||
hf_transfer
|
|
||||||
ray
|
ray
|
||||||
pandas
|
pandas
|
||||||
pyarrow
|
pyarrow
|
||||||
@@ -8,3 +7,4 @@ packaging
|
|||||||
typing-extensions==4.7.1
|
typing-extensions==4.7.1
|
||||||
pydantic
|
pydantic
|
||||||
pydantic-settings
|
pydantic-settings
|
||||||
|
hf-transfer
|
||||||
@@ -0,0 +1,65 @@
|
|||||||
|
variable "PUSH" {
|
||||||
|
default = "true"
|
||||||
|
}
|
||||||
|
|
||||||
|
variable "REPOSITORY" {
|
||||||
|
default = "runpod"
|
||||||
|
}
|
||||||
|
|
||||||
|
variable "BASE_IMAGE_VERSION" {
|
||||||
|
default = "1.0.0"
|
||||||
|
}
|
||||||
|
|
||||||
|
group "all" {
|
||||||
|
targets = ["base", "main"]
|
||||||
|
}
|
||||||
|
|
||||||
|
group "base" {
|
||||||
|
targets = ["base-1180", "base-1210"]
|
||||||
|
}
|
||||||
|
|
||||||
|
group "main" {
|
||||||
|
targets = ["worker-1180", "worker-1210"]
|
||||||
|
}
|
||||||
|
|
||||||
|
target "base-1180" {
|
||||||
|
tags = ["${REPOSITORY}/worker-vllm:base-${BASE_IMAGE_VERSION}-cuda11.8.0"]
|
||||||
|
context = "vllm-base-image"
|
||||||
|
dockerfile = "Dockerfile"
|
||||||
|
args = {
|
||||||
|
WORKER_CUDA_VERSION = "11.8.0"
|
||||||
|
}
|
||||||
|
output = ["type=docker,push=${PUSH}"]
|
||||||
|
}
|
||||||
|
|
||||||
|
target "base-1210" {
|
||||||
|
tags = ["${REPOSITORY}/worker-vllm:base-${BASE_IMAGE_VERSION}-cuda12.1.0"]
|
||||||
|
context = "vllm-base-image"
|
||||||
|
dockerfile = "Dockerfile"
|
||||||
|
args = {
|
||||||
|
WORKER_CUDA_VERSION = "12.1.0"
|
||||||
|
}
|
||||||
|
output = ["type=docker,push=${PUSH}"]
|
||||||
|
}
|
||||||
|
|
||||||
|
target "worker-1180" {
|
||||||
|
tags = ["${REPOSITORY}/worker-vllm:${BASE_IMAGE_VERSION}-cuda11.8.0"]
|
||||||
|
context = "."
|
||||||
|
dockerfile = "Dockerfile"
|
||||||
|
args = {
|
||||||
|
BASE_IMAGE_VERSION = "${BASE_IMAGE_VERSION}"
|
||||||
|
WORKER_CUDA_VERSION = "11.8.0"
|
||||||
|
}
|
||||||
|
output = ["type=docker,push=${PUSH}"]
|
||||||
|
}
|
||||||
|
|
||||||
|
target "worker-1210" {
|
||||||
|
tags = ["${REPOSITORY}/worker-vllm:${BASE_IMAGE_VERSION}-cuda12.1.0"]
|
||||||
|
context = "."
|
||||||
|
dockerfile = "Dockerfile"
|
||||||
|
args = {
|
||||||
|
BASE_IMAGE_VERSION = "${BASE_IMAGE_VERSION}"
|
||||||
|
WORKER_CUDA_VERSION = "12.1.0"
|
||||||
|
}
|
||||||
|
output = ["type=docker,push=${PUSH}"]
|
||||||
|
}
|
||||||
Binary file not shown.
|
After Width: | Height: | Size: 27 MiB |
+32
-23
@@ -1,29 +1,31 @@
|
|||||||
import os
|
import os
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
from torch.cuda import device_count
|
from torch.cuda import device_count
|
||||||
import os
|
from utils import get_int_bool_env
|
||||||
|
|
||||||
class EngineConfig:
|
class EngineConfig:
|
||||||
def __init__(self):
|
def __init__(self):
|
||||||
load_dotenv()
|
load_dotenv()
|
||||||
self.model_name_or_path, self.hf_home, self.model_revision = self._get_local_or_env("/local_model_path.txt", "MODEL_NAME")
|
self.hf_home = os.getenv("HF_HOME")
|
||||||
self.tokenizer_name_or_path, _, self.tokenizer_revision = self._get_local_or_env("/local_tokenizer_path.txt", "TOKENIZER_NAME")
|
# Check if /local_metadata.json exists
|
||||||
self.tokenizer_name_or_path = self.tokenizer_name_or_path or self.model_name_or_path
|
local_metadata = {}
|
||||||
self.quantization = self._get_quantization()
|
if os.path.exists("/local_metadata.json"):
|
||||||
self.config = self._initialize_config()
|
with open("/local_metadata.json", "r") as f:
|
||||||
|
local_metadata = json.load(f)
|
||||||
def _get_local_or_env(self, local_path, env_var):
|
if local_metadata.get("model_name") is None:
|
||||||
if os.path.exists(local_path):
|
raise ValueError("Model name is not found in /local_metadata.json, there was a problem when you baked the model in.")
|
||||||
|
logging.info("Using baked-in model")
|
||||||
os.environ["TRANSFORMERS_OFFLINE"] = "1"
|
os.environ["TRANSFORMERS_OFFLINE"] = "1"
|
||||||
os.environ["HF_HUB_OFFLINE"] = "1"
|
os.environ["HF_HUB_OFFLINE"] = "1"
|
||||||
with open(local_path, "r") as file:
|
|
||||||
return file.read().strip(), None, None
|
|
||||||
return os.getenv(env_var), os.getenv("HF_HOME"), os.getenv(f"{env_var.split('_')[0]}_REVISION") or None
|
|
||||||
|
|
||||||
def _get_quantization(self):
|
|
||||||
quantization = os.getenv("QUANTIZATION", "").lower()
|
|
||||||
return quantization if quantization in ["awq", "squeezellm", "gptq"] else None
|
|
||||||
|
|
||||||
|
self.model_name_or_path = local_metadata.get("model_name", os.getenv("MODEL_NAME"))
|
||||||
|
self.model_revision = local_metadata.get("revision", os.getenv("MODEL_REVISION"))
|
||||||
|
self.tokenizer_name_or_path = local_metadata.get("tokenizer_name", os.getenv("TOKENIZER_NAME")) or self.model_name_or_path
|
||||||
|
self.tokenizer_revision = local_metadata.get("tokenizer_revision", os.getenv("TOKENIZER_REVISION"))
|
||||||
|
self.quantization = local_metadata.get("quantization", os.getenv("QUANTIZATION"))
|
||||||
|
self.config = self._initialize_config()
|
||||||
def _initialize_config(self):
|
def _initialize_config(self):
|
||||||
args = {
|
args = {
|
||||||
"model": self.model_name_or_path,
|
"model": self.model_name_or_path,
|
||||||
@@ -34,9 +36,9 @@ class EngineConfig:
|
|||||||
"dtype": os.getenv("DTYPE", "half" if self.quantization else "auto"),
|
"dtype": os.getenv("DTYPE", "half" if self.quantization else "auto"),
|
||||||
"tokenizer": self.tokenizer_name_or_path,
|
"tokenizer": self.tokenizer_name_or_path,
|
||||||
"tokenizer_revision": self.tokenizer_revision,
|
"tokenizer_revision": self.tokenizer_revision,
|
||||||
"disable_log_stats": bool(int(os.getenv("DISABLE_LOG_STATS", 1))),
|
"disable_log_stats": get_int_bool_env("DISABLE_LOG_STATS", True),
|
||||||
"disable_log_requests": bool(int(os.getenv("DISABLE_LOG_REQUESTS", 1))),
|
"disable_log_requests": get_int_bool_env("DISABLE_LOG_REQUESTS", True),
|
||||||
"trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))),
|
"trust_remote_code": get_int_bool_env("TRUST_REMOTE_CODE", False),
|
||||||
"gpu_memory_utilization": float(os.getenv("GPU_MEMORY_UTILIZATION", 0.95)),
|
"gpu_memory_utilization": float(os.getenv("GPU_MEMORY_UTILIZATION", 0.95)),
|
||||||
"max_parallel_loading_workers": None if device_count() > 1 or not os.getenv("MAX_PARALLEL_LOADING_WORKERS") else int(os.getenv("MAX_PARALLEL_LOADING_WORKERS")),
|
"max_parallel_loading_workers": None if device_count() > 1 or not os.getenv("MAX_PARALLEL_LOADING_WORKERS") else int(os.getenv("MAX_PARALLEL_LOADING_WORKERS")),
|
||||||
"max_model_len": int(os.getenv("MAX_MODEL_LEN")) if os.getenv("MAX_MODEL_LEN") else None,
|
"max_model_len": int(os.getenv("MAX_MODEL_LEN")) if os.getenv("MAX_MODEL_LEN") else None,
|
||||||
@@ -45,9 +47,16 @@ class EngineConfig:
|
|||||||
"kv_cache_dtype": os.getenv("KV_CACHE_DTYPE"),
|
"kv_cache_dtype": os.getenv("KV_CACHE_DTYPE"),
|
||||||
"block_size": int(os.getenv("BLOCK_SIZE")) if os.getenv("BLOCK_SIZE") else None,
|
"block_size": int(os.getenv("BLOCK_SIZE")) if os.getenv("BLOCK_SIZE") else None,
|
||||||
"swap_space": int(os.getenv("SWAP_SPACE")) if os.getenv("SWAP_SPACE") else None,
|
"swap_space": int(os.getenv("SWAP_SPACE")) if os.getenv("SWAP_SPACE") else None,
|
||||||
"max_context_len_to_capture": int(os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE")) if os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE") else None,
|
"max_seq_len_to_capture": int(os.getenv("MAX_SEQ_LEN_TO_CAPTURE")) if os.getenv("MAX_SEQ_LEN_TO_CAPTURE") else None,
|
||||||
"disable_custom_all_reduce": bool(int(os.getenv("DISABLE_CUSTOM_ALL_REDUCE", 0))),
|
"disable_custom_all_reduce": get_int_bool_env("DISABLE_CUSTOM_ALL_REDUCE", False),
|
||||||
"enforce_eager": bool(int(os.getenv("ENFORCE_EAGER", 0)))
|
"enforce_eager": get_int_bool_env("ENFORCE_EAGER", False)
|
||||||
}
|
}
|
||||||
|
if args["kv_cache_dtype"] == "fp8_e5m2":
|
||||||
|
args["kv_cache_dtype"] = "fp8"
|
||||||
|
logging.warning("Using fp8_e5m2 is deprecated. Please use fp8 instead.")
|
||||||
|
if os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE"):
|
||||||
|
args["max_seq_len_to_capture"] = int(os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE"))
|
||||||
|
logging.warning("Using MAX_CONTEXT_LEN_TO_CAPTURE is deprecated. Please use MAX_SEQ_LEN_TO_CAPTURE instead.")
|
||||||
|
|
||||||
return {k: v for k, v in args.items() if v is not None}
|
|
||||||
|
return {k: v for k, v in args.items() if v not in [None, ""]}
|
||||||
|
|||||||
@@ -0,0 +1,27 @@
|
|||||||
|
import os
|
||||||
|
from huggingface_hub import snapshot_download
|
||||||
|
import json
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
model_name = os.getenv("MODEL_NAME")
|
||||||
|
if not model_name:
|
||||||
|
raise ValueError("Must specify model name by adding --build-arg MODEL_NAME=<your model's repo>")
|
||||||
|
revision = os.getenv("MODEL_REVISION") or None
|
||||||
|
snapshot_download(model_name, revision=revision, cache_dir=os.getenv("HF_HOME"))
|
||||||
|
|
||||||
|
tokenizer_name = os.getenv("TOKENIZER_NAME") or None
|
||||||
|
tokenizer_revision = os.getenv("TOKENIZER_REVISION") or None
|
||||||
|
if tokenizer_name:
|
||||||
|
snapshot_download(tokenizer_name, revision=tokenizer_revision, cache_dir=os.getenv("HF_HOME"))
|
||||||
|
|
||||||
|
# Create file with metadata of baked in model and/or tokenizer
|
||||||
|
|
||||||
|
with open("/local_metadata.json", "w") as f:
|
||||||
|
json.dump({
|
||||||
|
"model_name": model_name,
|
||||||
|
"revision": revision,
|
||||||
|
"tokenizer_name": tokenizer_name or model_name,
|
||||||
|
"tokenizer_revision": tokenizer_revision or revision,
|
||||||
|
"quantization": os.getenv("QUANTIZATION")
|
||||||
|
}, f)
|
||||||
|
|
||||||
+9
-1
@@ -5,6 +5,7 @@ import json
|
|||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
from torch.cuda import device_count
|
from torch.cuda import device_count
|
||||||
from typing import AsyncGenerator
|
from typing import AsyncGenerator
|
||||||
|
import time
|
||||||
|
|
||||||
from vllm import AsyncLLMEngine, AsyncEngineArgs
|
from vllm import AsyncLLMEngine, AsyncEngineArgs
|
||||||
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
|
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
|
||||||
@@ -100,7 +101,11 @@ class vLLMEngine:
|
|||||||
|
|
||||||
def _initialize_llm(self):
|
def _initialize_llm(self):
|
||||||
try:
|
try:
|
||||||
return AsyncLLMEngine.from_engine_args(AsyncEngineArgs(**self.config))
|
start = time.time()
|
||||||
|
engine = AsyncLLMEngine.from_engine_args(AsyncEngineArgs(**self.config))
|
||||||
|
end = time.time()
|
||||||
|
logging.info(f"Initialized vLLM engine in {end - start:.2f}s")
|
||||||
|
return engine
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logging.error("Error initializing vLLM engine: %s", e)
|
logging.error("Error initializing vLLM engine: %s", e)
|
||||||
raise e
|
raise e
|
||||||
@@ -136,6 +141,9 @@ class OpenAIvLLMEngine:
|
|||||||
|
|
||||||
async def _handle_model_request(self):
|
async def _handle_model_request(self):
|
||||||
models = await self.chat_engine.show_available_models()
|
models = await self.chat_engine.show_available_models()
|
||||||
|
fixed_model = models.data[0]
|
||||||
|
fixed_model.id = self.served_model_name
|
||||||
|
models.data = [fixed_model]
|
||||||
return models.model_dump()
|
return models.model_dump()
|
||||||
|
|
||||||
async def _handle_chat_or_completion_request(self, openai_request: JobInput):
|
async def _handle_chat_or_completion_request(self, openai_request: JobInput):
|
||||||
|
|||||||
+6
-1
@@ -1,6 +1,6 @@
|
|||||||
|
import os
|
||||||
import logging
|
import logging
|
||||||
from http import HTTPStatus
|
from http import HTTPStatus
|
||||||
from typing import Any, Dict
|
|
||||||
from vllm.utils import random_uuid
|
from vllm.utils import random_uuid
|
||||||
from vllm.entrypoints.openai.protocol import ErrorResponse
|
from vllm.entrypoints.openai.protocol import ErrorResponse
|
||||||
from vllm import SamplingParams
|
from vllm import SamplingParams
|
||||||
@@ -65,4 +65,9 @@ def create_error_response(message: str, err_type: str = "BadRequestError", statu
|
|||||||
type=err_type,
|
type=err_type,
|
||||||
code=status_code.value)
|
code=status_code.value)
|
||||||
|
|
||||||
|
def get_int_bool_env(env_var: str, default: bool) -> bool:
|
||||||
|
return int(os.getenv(env_var, int(default))) == 1
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
+82
-21
@@ -20,16 +20,22 @@ RUN apt-get update -y \
|
|||||||
# Set working directory
|
# Set working directory
|
||||||
WORKDIR /vllm-installation
|
WORKDIR /vllm-installation
|
||||||
|
|
||||||
|
RUN ldconfig /usr/local/cuda-$(echo "$WORKER_CUDA_VERSION" | sed 's/\.0$//')/compat/
|
||||||
|
|
||||||
# Install build and runtime dependencies
|
# Install build and runtime dependencies
|
||||||
COPY vllm/requirements-${WORKER_CUDA_VERSION}.txt requirements.txt
|
COPY vllm/requirements-common.txt requirements-common.txt
|
||||||
|
COPY vllm/requirements-cuda${WORKER_CUDA_VERSION}.txt requirements-cuda.txt
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
pip install -r requirements.txt
|
pip install -r requirements-cuda.txt
|
||||||
|
|
||||||
# Install development dependencies
|
# Install development dependencies
|
||||||
COPY vllm/requirements-dev.txt requirements-dev.txt
|
COPY vllm/requirements-dev.txt requirements-dev.txt
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
pip install -r requirements-dev.txt
|
pip install -r requirements-dev.txt
|
||||||
|
|
||||||
|
ARG torch_cuda_arch_list='7.0 7.5 8.0 8.6 8.9 9.0+PTX'
|
||||||
|
ENV TORCH_CUDA_ARCH_LIST=${torch_cuda_arch_list}
|
||||||
|
|
||||||
FROM dev AS build
|
FROM dev AS build
|
||||||
|
|
||||||
# Re-declare ARG after FROM
|
# Re-declare ARG after FROM
|
||||||
@@ -40,26 +46,69 @@ COPY vllm/requirements-build.txt requirements-build.txt
|
|||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
pip install -r requirements-build.txt
|
pip install -r requirements-build.txt
|
||||||
|
|
||||||
|
# install compiler cache to speed up compilation leveraging local or remote caching
|
||||||
|
RUN apt-get update -y && apt-get install -y ccache
|
||||||
|
|
||||||
# Copy necessary files
|
# Copy necessary files
|
||||||
COPY vllm/csrc csrc
|
COPY vllm/csrc csrc
|
||||||
COPY vllm/setup.py setup.py
|
COPY vllm/setup.py setup.py
|
||||||
|
COPY vllm/cmake cmake
|
||||||
|
COPY vllm/CMakeLists.txt CMakeLists.txt
|
||||||
|
COPY vllm/requirements-common.txt requirements-common.txt
|
||||||
|
COPY vllm/requirements-cuda${WORKER_CUDA_VERSION}.txt requirements-cuda.txt
|
||||||
COPY vllm/pyproject.toml pyproject.toml
|
COPY vllm/pyproject.toml pyproject.toml
|
||||||
COPY vllm/vllm/__init__.py vllm/__init__.py
|
COPY vllm/vllm vllm
|
||||||
|
|
||||||
# Set environment variables for building extensions
|
# Set environment variables for building extensions
|
||||||
ARG torch_cuda_arch_list='7.0 7.5 8.0 8.6 8.9 9.0+PTX'
|
|
||||||
ENV TORCH_CUDA_ARCH_LIST=${torch_cuda_arch_list}
|
|
||||||
ARG max_jobs=48
|
|
||||||
ENV MAX_JOBS=${max_jobs}
|
|
||||||
ARG nvcc_threads=1024
|
|
||||||
ENV NVCC_THREADS=${nvcc_threads}
|
|
||||||
ENV WORKER_CUDA_VERSION=${WORKER_CUDA_VERSION}
|
ENV WORKER_CUDA_VERSION=${WORKER_CUDA_VERSION}
|
||||||
ENV VLLM_INSTALL_PUNICA_KERNELS=0
|
ENV VLLM_INSTALL_PUNICA_KERNELS=0
|
||||||
# Build extensions
|
# Build extensions
|
||||||
RUN ldconfig /usr/local/cuda-$(echo "$WORKER_CUDA_VERSION" | sed 's/\.0$//')/compat/
|
ENV CCACHE_DIR=/root/.cache/ccache
|
||||||
RUN python3 setup.py build_ext --inplace
|
RUN --mount=type=cache,target=/root/.cache/ccache \
|
||||||
|
--mount=type=cache,target=/root/.cache/pip \
|
||||||
|
python3 setup.py bdist_wheel --dist-dir=dist
|
||||||
|
|
||||||
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-runtime-ubuntu22.04 AS vllm-base
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
|
pip cache remove vllm_nccl*
|
||||||
|
|
||||||
|
FROM dev as flash-attn-builder
|
||||||
|
# max jobs used for build
|
||||||
|
# flash attention version
|
||||||
|
ARG flash_attn_version=v2.5.8
|
||||||
|
ENV FLASH_ATTN_VERSION=${flash_attn_version}
|
||||||
|
|
||||||
|
WORKDIR /usr/src/flash-attention-v2
|
||||||
|
|
||||||
|
# Download the wheel or build it if a pre-compiled release doesn't exist
|
||||||
|
RUN pip --verbose wheel flash-attn==${FLASH_ATTN_VERSION} \
|
||||||
|
--no-build-isolation --no-deps --no-cache-dir
|
||||||
|
|
||||||
|
FROM dev as NCCL-installer
|
||||||
|
|
||||||
|
# Re-declare ARG after FROM
|
||||||
|
ARG WORKER_CUDA_VERSION
|
||||||
|
|
||||||
|
# Update and install necessary libraries
|
||||||
|
RUN apt-get update -y \
|
||||||
|
&& apt-get install -y wget
|
||||||
|
|
||||||
|
# Install NCCL library
|
||||||
|
RUN if [ "$WORKER_CUDA_VERSION" = "11.8.0" ]; then \
|
||||||
|
wget https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.0-1_all.deb \
|
||||||
|
&& dpkg -i cuda-keyring_1.0-1_all.deb \
|
||||||
|
&& apt-get update \
|
||||||
|
&& apt install -y libnccl2=2.15.5-1+cuda11.8 libnccl-dev=2.15.5-1+cuda11.8; \
|
||||||
|
elif [ "$WORKER_CUDA_VERSION" = "12.1.0" ]; then \
|
||||||
|
wget https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.0-1_all.deb \
|
||||||
|
&& dpkg -i cuda-keyring_1.0-1_all.deb \
|
||||||
|
&& apt-get update \
|
||||||
|
&& apt install -y libnccl2=2.17.1-1+cuda12.1 libnccl-dev=2.17.1-1+cuda12.1; \
|
||||||
|
else \
|
||||||
|
echo "Unsupported CUDA version: $WORKER_CUDA_VERSION"; \
|
||||||
|
exit 1; \
|
||||||
|
fi
|
||||||
|
|
||||||
|
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-base-ubuntu22.04 AS vllm-base
|
||||||
|
|
||||||
# Re-declare ARG after FROM
|
# Re-declare ARG after FROM
|
||||||
ARG WORKER_CUDA_VERSION
|
ARG WORKER_CUDA_VERSION
|
||||||
@@ -69,20 +118,32 @@ RUN apt-get update -y \
|
|||||||
&& apt-get install -y python3-pip
|
&& apt-get install -y python3-pip
|
||||||
|
|
||||||
# Set working directory
|
# Set working directory
|
||||||
WORKDIR /vllm-installation
|
WORKDIR /vllm-workspace
|
||||||
|
|
||||||
|
RUN ldconfig /usr/local/cuda-$(echo "$WORKER_CUDA_VERSION" | sed 's/\.0$//')/compat/
|
||||||
|
|
||||||
# Install runtime dependencies
|
RUN --mount=type=bind,from=build,src=/vllm-installation/dist,target=/vllm-workspace/dist \
|
||||||
COPY vllm/requirements-${WORKER_CUDA_VERSION}.txt requirements.txt
|
--mount=type=cache,target=/root/.cache/pip \
|
||||||
|
pip install dist/*.whl --verbose
|
||||||
|
|
||||||
|
RUN --mount=type=bind,from=flash-attn-builder,src=/usr/src/flash-attention-v2,target=/usr/src/flash-attention-v2 \
|
||||||
|
--mount=type=cache,target=/root/.cache/pip \
|
||||||
|
pip install /usr/src/flash-attention-v2/*.whl --no-cache-dir
|
||||||
|
|
||||||
|
FROM vllm-base AS runtime
|
||||||
|
|
||||||
|
# install additional dependencies for openai api server
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
pip install -r requirements.txt
|
pip install accelerate hf_transfer modelscope tensorizer
|
||||||
|
|
||||||
# Copy built files from the build stage
|
|
||||||
COPY --from=build /vllm-installation/vllm/*.so /vllm-installation/vllm/
|
|
||||||
COPY vllm/vllm vllm
|
|
||||||
|
|
||||||
# Set PYTHONPATH environment variable
|
# Set PYTHONPATH environment variable
|
||||||
ENV PYTHONPATH="/"
|
ENV PYTHONPATH="/"
|
||||||
|
|
||||||
|
# Copy NCCL library
|
||||||
|
COPY --from=NCCL-installer /usr/lib/x86_64-linux-gnu/libnccl.so.2 /usr/lib/x86_64-linux-gnu/libnccl.so.2
|
||||||
|
# Set the VLLM_NCCL_SO_PATH environment variable
|
||||||
|
ENV VLLM_NCCL_SO_PATH="/usr/lib/x86_64-linux-gnu/libnccl.so.2"
|
||||||
|
|
||||||
|
|
||||||
# Validate the installation
|
# Validate the installation
|
||||||
RUN python3 -c "import sys; print(sys.path); import vllm; print(vllm.__file__)"
|
RUN python3 -c "import vllm; print(vllm.__file__)"
|
||||||
+1
-1
Submodule vllm-base-image/vllm updated: c46d230a62...ba8f5e79e1
@@ -0,0 +1,2 @@
|
|||||||
|
version: '0.4.2'
|
||||||
|
dev_version: '0.4.2'
|
||||||
Reference in New Issue
Block a user