Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c8458fef2b | ||
|
|
bad5ddd892 | ||
|
|
f19ce12ab0 | ||
|
|
1bb6f84541 | ||
|
|
30cb56a3df | ||
|
|
00add8707a | ||
|
|
ec7ea0b760 | ||
|
|
9f2cb7b1d0 | ||
|
|
4abe494635 | ||
|
|
4f61b04afe | ||
|
|
874379a0c5 | ||
|
|
f06a64d5b9 | ||
|
|
0a5b5bc095 | ||
|
|
2936e4d95d | ||
|
|
cee4e484d5 | ||
|
|
6160769996 | ||
|
|
d25b6f9628 | ||
|
|
c8ee100d80 | ||
|
|
fee8d8eee4 | ||
|
|
db7167d57f | ||
|
|
d91ccb866f | ||
|
|
36e9b670ee |
@@ -1,45 +0,0 @@
|
|||||||
name: CD | Docker-Build-Release
|
|
||||||
|
|
||||||
on:
|
|
||||||
push:
|
|
||||||
branches:
|
|
||||||
- "main"
|
|
||||||
release:
|
|
||||||
types: [published]
|
|
||||||
workflow_dispatch:
|
|
||||||
inputs:
|
|
||||||
image_tag:
|
|
||||||
description: "Docker Image Tag"
|
|
||||||
required: false
|
|
||||||
default: "dev"
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
docker-build:
|
|
||||||
runs-on: DO
|
|
||||||
# DO is a custom runner deployed on DigitalOcean, only available for workflows under the runpod-workers organization.
|
|
||||||
# If you would like to use this workflow, you can replace DO with ubuntu-latest or any other runner.
|
|
||||||
|
|
||||||
strategy:
|
|
||||||
matrix:
|
|
||||||
cuda_version: [11.8.0, 12.1.0]
|
|
||||||
|
|
||||||
steps:
|
|
||||||
- name: Set up QEMU
|
|
||||||
uses: docker/setup-qemu-action@v2
|
|
||||||
|
|
||||||
- name: Set up Docker Buildx
|
|
||||||
uses: docker/setup-buildx-action@v2
|
|
||||||
|
|
||||||
- name: Login to Docker Hub
|
|
||||||
uses: docker/login-action@v2
|
|
||||||
with:
|
|
||||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
|
||||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
|
||||||
|
|
||||||
# Build and push step
|
|
||||||
- name: Build and push
|
|
||||||
uses: docker/build-push-action@v4
|
|
||||||
with:
|
|
||||||
push: true
|
|
||||||
tags: ${{ vars.DOCKERHUB_REPO }}/${{ vars.DOCKERHUB_IMG }}:${{ (github.event_name == 'release' && github.event.release.tag_name) || (github.event_name == 'workflow_dispatch' && github.event.inputs.image_tag) || 'dev' }}-cuda${{ matrix.cuda_version }}
|
|
||||||
build-args: WORKER_CUDA_VERSION=${{ matrix.cuda_version }}
|
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
[submodule "vllm-base-image/vllm"]
|
||||||
|
path = vllm-base-image/vllm
|
||||||
|
url = https://github.com/runpod/vllm-fork-for-sls-worker.git
|
||||||
+8
-6
@@ -1,5 +1,6 @@
|
|||||||
ARG WORKER_CUDA_VERSION=11.8.0
|
ARG WORKER_CUDA_VERSION=11.8.0
|
||||||
FROM runpod/worker-vllm:base-0.3.0-cuda${WORKER_CUDA_VERSION} AS vllm-base
|
ARG BASE_IMAGE_VERSION=1.0.0
|
||||||
|
FROM runpod/worker-vllm:base-${BASE_IMAGE_VERSION}-cuda${WORKER_CUDA_VERSION} AS vllm-base
|
||||||
|
|
||||||
RUN apt-get update -y \
|
RUN apt-get update -y \
|
||||||
&& apt-get install -y python3-pip
|
&& apt-get install -y python3-pip
|
||||||
@@ -19,7 +20,7 @@ ARG MODEL_REVISION=""
|
|||||||
ARG TOKENIZER_REVISION=""
|
ARG TOKENIZER_REVISION=""
|
||||||
|
|
||||||
ENV MODEL_NAME=$MODEL_NAME \
|
ENV MODEL_NAME=$MODEL_NAME \
|
||||||
MODEL_REVISION=$REVISION \
|
MODEL_REVISION=$MODEL_REVISION \
|
||||||
TOKENIZER_NAME=$TOKENIZER_NAME \
|
TOKENIZER_NAME=$TOKENIZER_NAME \
|
||||||
TOKENIZER_REVISION=$TOKENIZER_REVISION \
|
TOKENIZER_REVISION=$TOKENIZER_REVISION \
|
||||||
BASE_PATH=$BASE_PATH \
|
BASE_PATH=$BASE_PATH \
|
||||||
@@ -27,11 +28,11 @@ ENV MODEL_NAME=$MODEL_NAME \
|
|||||||
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
|
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
|
||||||
HUGGINGFACE_HUB_CACHE="${BASE_PATH}/huggingface-cache/hub" \
|
HUGGINGFACE_HUB_CACHE="${BASE_PATH}/huggingface-cache/hub" \
|
||||||
HF_HOME="${BASE_PATH}/huggingface-cache/hub" \
|
HF_HOME="${BASE_PATH}/huggingface-cache/hub" \
|
||||||
HF_TRANSFER=1
|
HF_HUB_ENABLE_HF_TRANSFER=1
|
||||||
|
|
||||||
ENV PYTHONPATH="/:/vllm-installation"
|
ENV PYTHONPATH="/:/vllm-workspace"
|
||||||
|
|
||||||
COPY builder/download_model.py /download_model.py
|
COPY src/download_model.py /download_model.py
|
||||||
RUN --mount=type=secret,id=HF_TOKEN,required=false \
|
RUN --mount=type=secret,id=HF_TOKEN,required=false \
|
||||||
if [ -f /run/secrets/HF_TOKEN ]; then \
|
if [ -f /run/secrets/HF_TOKEN ]; then \
|
||||||
export HF_TOKEN=$(cat /run/secrets/HF_TOKEN); \
|
export HF_TOKEN=$(cat /run/secrets/HF_TOKEN); \
|
||||||
@@ -42,7 +43,8 @@ RUN --mount=type=secret,id=HF_TOKEN,required=false \
|
|||||||
|
|
||||||
# Add source files
|
# Add source files
|
||||||
COPY src /src
|
COPY src /src
|
||||||
|
# Remove download_model.py
|
||||||
|
RUN rm /download_model.py
|
||||||
|
|
||||||
# Start the handler
|
# Start the handler
|
||||||
CMD ["python3", "/src/handler.py"]
|
CMD ["python3", "/src/handler.py"]
|
||||||
@@ -1,25 +1,34 @@
|
|||||||
<div align="center">
|
<div align="center">
|
||||||
|
|
||||||
<h1> vLLM Serverless Endpoint Worker </h1>
|
# OpenAI-Compatible vLLM Serverless Endpoint Worker
|
||||||
|
Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https://github.com/vllm-project/vllm) Inference Engine on RunPod Serverless with just a few clicks.
|
||||||
|
<!--
|
||||||
|

|
||||||
|

|
||||||
|
\
|
||||||
|
 -->
|
||||||
|
<!--
|
||||||
|
 -->
|
||||||
|
|
||||||
[](https://github.com/runpod-workers/worker-vllm/actions/workflows/docker-build-release.yml)
|
|
||||||
|
|
||||||
Deploy Blazing-fast LLMs powered by [vLLM](https://github.com/vllm-project/vllm) on RunPod Serverless in a few clicks.
|
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
### Worker vLLM 0.3.0: What's New since 0.2.0:
|
# News:
|
||||||
- **🚀 Full OpenAI Compatibility 🚀**
|
|
||||||
|
### 1. UI for Deploying vLLM Worker on RunPod console:
|
||||||
|

|
||||||
|
|
||||||
|
### 2. Worker vLLM `1.0.0` with vLLM `0.4.2` now available under `stable` tags
|
||||||
|
Update 1.0.0 is now available, use the image tag `runpod/worker-vllm:stable-cuda12.1.0` or `runpod/worker-vllm:stable-cuda11.8.0`.
|
||||||
|
|
||||||
|
### 3. OpenAI-Compatible [Embedding Worker](https://github.com/runpod-workers/worker-infinity-embedding) Released
|
||||||
|
Deploy your own OpenAI-compatible Serverless Endpoint on RunPod with multiple embedding models and fast inference for RAG and more!
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
### 4. Caching Accross RunPod Machines
|
||||||
|
Worker vLLM is now cached on all RunPod machines, resulting in near-instant deployment! Previously, downloading and extracting the image took 3-5 minutes on average.
|
||||||
|
|
||||||
You may now use your deployment with any OpenAI Codebase by changing **only 3 lines** in total. The supported routes are <ins>Chat Completions</ins>, <ins>Completions</ins>, and <ins>Models</ins> - with both streaming and non-streaming.
|
|
||||||
- **Dynamic Batch Size** - time-to-first token as fast no batching, while maintaining the performance of batched token streaming throughout the request.
|
|
||||||
- vLLM 0.2.7 -> 0.3.2
|
|
||||||
- Gemma, DeepSeek MoE and OLMo support.
|
|
||||||
- FP8 KV Cache support
|
|
||||||
- New supported parameters
|
|
||||||
- We're working on adding support for Multi-LoRA ⚙️
|
|
||||||
- Support for a wide range of new settings for your endpoint, such as Custom chat templates.
|
|
||||||
- Fixed Tensor Parallelism, baking model into images, and more bugs.
|
|
||||||
- Refactors and general improvements.
|
|
||||||
|
|
||||||
## Table of Contents
|
## Table of Contents
|
||||||
- [Setting up the Serverless Worker](#setting-up-the-serverless-worker)
|
- [Setting up the Serverless Worker](#setting-up-the-serverless-worker)
|
||||||
@@ -53,8 +62,11 @@ Deploy Blazing-fast LLMs powered by [vLLM](https://github.com/vllm-project/vllm)
|
|||||||
# Setting up the Serverless Worker
|
# Setting up the Serverless Worker
|
||||||
|
|
||||||
### Option 1: Deploy Any Model Using Pre-Built Docker Image [Recommended]
|
### Option 1: Deploy Any Model Using Pre-Built Docker Image [Recommended]
|
||||||
> [!TIP]
|
|
||||||
> This is the recommended way to deploy your model, as it does not require you to build a Docker image, upload heavy models to DockerHub and wait for workers to download them. Instead, use this option to deploy your model in a few clicks. For even more convenience, attach a network storage volume to your Endpoint, which will download the model once and share it across all workers.
|
> [!NOTE]
|
||||||
|
> You can now deploy from the dedicated UI on the RunPod console with all of the settings and choices listed.
|
||||||
|
> Try now by accessing in Explore or Serverless pages on the RunPod console!
|
||||||
|
|
||||||
|
|
||||||
We now offer a pre-built Docker Image for the vLLM Worker that you can configure entirely with Environment Variables when creating the RunPod Serverless Endpoint:
|
We now offer a pre-built Docker Image for the vLLM Worker that you can configure entirely with Environment Variables when creating the RunPod Serverless Endpoint:
|
||||||
|
|
||||||
@@ -66,17 +78,17 @@ Below is a summary of the available RunPod Worker images, categorized by image s
|
|||||||
|
|
||||||
| CUDA Version | Stable Image Tag | Development Image Tag | Note |
|
| CUDA Version | Stable Image Tag | Development Image Tag | Note |
|
||||||
|--------------|-----------------------------------|-----------------------------------|----------------------------------------------------------------------|
|
|--------------|-----------------------------------|-----------------------------------|----------------------------------------------------------------------|
|
||||||
| 11.8.0 | `runpod/worker-vllm:0.3.0-cuda11.8.0` | `runpod/worker-vllm:dev-cuda11.8.0` | Available on all RunPod Workers without additional selection needed. |
|
| 11.8.0 | `runpod/worker-vllm:stable-cuda11.8.0` | `runpod/worker-vllm:dev-cuda11.8.0` | Available on all RunPod Workers without additional selection needed. |
|
||||||
| 12.1.0 | `runpod/worker-vllm:0.3.0-cuda12.1.0` | `runpod/worker-vllm:dev-cuda12.1.0` | When creating an Endpoint, select CUDA Version 12.2 and 12.1 in the filter. |
|
| 12.1.0 | `runpod/worker-vllm:stable-cuda12.1.0` | `runpod/worker-vllm:dev-cuda12.1.0` | When creating an Endpoint, select CUDA Version 12.3, 12.2 and 12.1 in the filter. |
|
||||||
|
|
||||||
|
|
||||||
This table provides a quick reference to the image tags you should use based on the desired CUDA version and image stability (Stable or Development). Ensure to follow the selection note for CUDA 12.1.0 compatibility.
|
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
#### Prerequisites
|
#### Prerequisites
|
||||||
- RunPod Account
|
- RunPod Account
|
||||||
|
|
||||||
#### Environment Variables
|
#### Environment Variables/Settings
|
||||||
> Note: `0` is equivalent to `False` and `1` is equivalent to `True` for boolean values.
|
> Note: `0` is equivalent to `False` and `1` is equivalent to `True` for boolean values.
|
||||||
|
|
||||||
| Name | Default | Type/Choices | Description |
|
| Name | Default | Type/Choices | Description |
|
||||||
@@ -84,14 +96,14 @@ This table provides a quick reference to the image tags you should use based on
|
|||||||
**LLM Settings**
|
**LLM Settings**
|
||||||
| `MODEL_NAME`**\*** | - | `str` | Hugging Face Model Repository (e.g., `openchat/openchat-3.5-1210`). |
|
| `MODEL_NAME`**\*** | - | `str` | Hugging Face Model Repository (e.g., `openchat/openchat-3.5-1210`). |
|
||||||
| `MODEL_REVISION` | `None` | `str` |Model revision(branch) to load. |
|
| `MODEL_REVISION` | `None` | `str` |Model revision(branch) to load. |
|
||||||
| `MAX_MODEL_LENGTH` | Model's maximum | `int` |Maximum number of tokens for the engine to handle per request. |
|
| `MAX_MODEL_LEN` | Model's maximum | `int` |Maximum number of tokens for the engine to handle per request. |
|
||||||
| `BASE_PATH` | `/runpod-volume` | `str` |Storage directory for Huggingface cache and model. Utilizes network storage if attached when pointed at `/runpod-volume`, which will have only one worker download the model once, which all workers will be able to load. If no network volume is present, creates a local directory within each worker. |
|
| `BASE_PATH` | `/runpod-volume` | `str` |Storage directory for Huggingface cache and model. Utilizes network storage if attached when pointed at `/runpod-volume`, which will have only one worker download the model once, which all workers will be able to load. If no network volume is present, creates a local directory within each worker. |
|
||||||
| `LOAD_FORMAT` | `auto` | `str` |Format to load model in. |
|
| `LOAD_FORMAT` | `auto` | `str` |Format to load model in. |
|
||||||
| `HF_TOKEN` | - | `str` |Hugging Face token for private and gated models. |
|
| `HF_TOKEN` | - | `str` |Hugging Face token for private and gated models. |
|
||||||
| `QUANTIZATION` | `None` | `awq`, `squeezellm`, `gptq` |Quantization of given model. The model must already be quantized. |
|
| `QUANTIZATION` | `None` | `awq`, `squeezellm`, `gptq` |Quantization of given model. The model must already be quantized. |
|
||||||
| `TRUST_REMOTE_CODE` | `0` | boolean as `int` |Trust remote code for Hugging Face models. Can help with Mixtral 8x7B, Quantized models, and unusual models/architectures.
|
| `TRUST_REMOTE_CODE` | `0` | boolean as `int` |Trust remote code for Hugging Face models. Can help with Mixtral 8x7B, Quantized models, and unusual models/architectures.
|
||||||
| `SEED` | `0` | `int` |Sets random seed for operations. |
|
| `SEED` | `0` | `int` |Sets random seed for operations. |
|
||||||
| `KV_CACHE_DTYPE` | `auto` | boolean as `int` |Data type for kv cache storage. Uses `DTYPE` if set to `auto`. |
|
| `KV_CACHE_DTYPE` | `auto` | `auto`, `fp8` |Data type for kv cache storage. Uses `DTYPE` if set to `auto`. |
|
||||||
| `DTYPE` | `auto` | `auto`, `half`, `float16`, `bfloat16`, `float`, `float32` |Sets datatype/precision for model weights and activations. |
|
| `DTYPE` | `auto` | `auto`, `half`, `float16`, `bfloat16`, `float`, `float32` |Sets datatype/precision for model weights and activations. |
|
||||||
**Tokenizer Settings**
|
**Tokenizer Settings**
|
||||||
| `TOKENIZER_NAME` | `None` | `str` |Tokenizer repository to use a different tokenizer than the model's default. |
|
| `TOKENIZER_NAME` | `None` | `str` |Tokenizer repository to use a different tokenizer than the model's default. |
|
||||||
@@ -170,6 +182,8 @@ Below are all supported model architectures (and examples of each) that you can
|
|||||||
- Baichuan & Baichuan2 (`baichuan-inc/Baichuan2-13B-Chat`, `baichuan-inc/Baichuan-7B`, etc.)
|
- Baichuan & Baichuan2 (`baichuan-inc/Baichuan2-13B-Chat`, `baichuan-inc/Baichuan-7B`, etc.)
|
||||||
- BLOOM (`bigscience/bloom`, `bigscience/bloomz`, etc.)
|
- BLOOM (`bigscience/bloom`, `bigscience/bloomz`, etc.)
|
||||||
- ChatGLM (`THUDM/chatglm2-6b`, `THUDM/chatglm3-6b`, etc.)
|
- ChatGLM (`THUDM/chatglm2-6b`, `THUDM/chatglm3-6b`, etc.)
|
||||||
|
- Command-R (`CohereForAI/c4ai-command-r-v01`, etc.)
|
||||||
|
- DBRX (`databricks/dbrx-base`, `databricks/dbrx-instruct` etc.)
|
||||||
- DeciLM (`Deci/DeciLM-7B`, `Deci/DeciLM-7B-instruct`, etc.)
|
- DeciLM (`Deci/DeciLM-7B`, `Deci/DeciLM-7B-instruct`, etc.)
|
||||||
- Falcon (`tiiuae/falcon-7b`, `tiiuae/falcon-40b`, `tiiuae/falcon-rw-7b`, etc.)
|
- Falcon (`tiiuae/falcon-7b`, `tiiuae/falcon-40b`, `tiiuae/falcon-rw-7b`, etc.)
|
||||||
- Gemma (`google/gemma-2b`, `google/gemma-7b`, etc.)
|
- Gemma (`google/gemma-2b`, `google/gemma-7b`, etc.)
|
||||||
@@ -179,16 +193,23 @@ Below are all supported model architectures (and examples of each) that you can
|
|||||||
- GPT-NeoX (`EleutherAI/gpt-neox-20b`, `databricks/dolly-v2-12b`, `stabilityai/stablelm-tuned-alpha-7b`, etc.)
|
- GPT-NeoX (`EleutherAI/gpt-neox-20b`, `databricks/dolly-v2-12b`, `stabilityai/stablelm-tuned-alpha-7b`, etc.)
|
||||||
- InternLM (`internlm/internlm-7b`, `internlm/internlm-chat-7b`, etc.)
|
- InternLM (`internlm/internlm-7b`, `internlm/internlm-chat-7b`, etc.)
|
||||||
- InternLM2 (`internlm/internlm2-7b`, `internlm/internlm2-chat-7b`, etc.)
|
- InternLM2 (`internlm/internlm2-7b`, `internlm/internlm2-chat-7b`, etc.)
|
||||||
- LLaMA & LLaMA-2 (`meta-llama/Llama-2-70b-hf`, `lmsys/vicuna-13b-v1.3`, `young-geng/koala`, `openlm-research/open_llama_13b`, etc.)
|
- Jais (`core42/jais-13b`, `core42/jais-13b-chat`, `core42/jais-30b-v3`, `core42/jais-30b-chat-v3`, etc.)
|
||||||
|
- LLaMA, Llama 2, and Meta Llama 3 (`meta-llama/Meta-Llama-3-8B-Instruct`, `meta-llama/Meta-Llama-3-70B-Instruct`, `meta-llama/Llama-2-70b-hf`, `lmsys/vicuna-13b-v1.3`, `young-geng/koala`, `openlm-research/open_llama_13b`, etc.)
|
||||||
|
- MiniCPM (`openbmb/MiniCPM-2B-sft-bf16`, `openbmb/MiniCPM-2B-dpo-bf16`, etc.)
|
||||||
- Mistral (`mistralai/Mistral-7B-v0.1`, `mistralai/Mistral-7B-Instruct-v0.1`, etc.)
|
- Mistral (`mistralai/Mistral-7B-v0.1`, `mistralai/Mistral-7B-Instruct-v0.1`, etc.)
|
||||||
- Mixtral (`mistralai/Mixtral-8x7B-v0.1`, `mistralai/Mixtral-8x7B-Instruct-v0.1`, etc.)
|
- Mixtral (`mistralai/Mixtral-8x7B-v0.1`, `mistralai/Mixtral-8x7B-Instruct-v0.1`, `mistral-community/Mixtral-8x22B-v0.1`, etc.)
|
||||||
- MPT (`mosaicml/mpt-7b`, `mosaicml/mpt-30b`, etc.)
|
- MPT (`mosaicml/mpt-7b`, `mosaicml/mpt-30b`, etc.)
|
||||||
- OLMo (`allenai/OLMo-1B`, `allenai/OLMo-7B`, etc.)
|
- OLMo (`allenai/OLMo-1B-hf`, `allenai/OLMo-7B-hf`, etc.)
|
||||||
- OPT (`facebook/opt-66b`, `facebook/opt-iml-max-30b`, etc.)
|
- OPT (`facebook/opt-66b`, `facebook/opt-iml-max-30b`, etc.)
|
||||||
|
- Orion (`OrionStarAI/Orion-14B-Base`, `OrionStarAI/Orion-14B-Chat`, etc.)
|
||||||
- Phi (`microsoft/phi-1_5`, `microsoft/phi-2`, etc.)
|
- Phi (`microsoft/phi-1_5`, `microsoft/phi-2`, etc.)
|
||||||
|
- Phi-3 (`microsoft/Phi-3-mini-4k-instruct`, `microsoft/Phi-3-mini-128k-instruct`, etc.)
|
||||||
- Qwen (`Qwen/Qwen-7B`, `Qwen/Qwen-7B-Chat`, etc.)
|
- Qwen (`Qwen/Qwen-7B`, `Qwen/Qwen-7B-Chat`, etc.)
|
||||||
- Qwen2 (`Qwen/Qwen2-7B-beta`, `Qwen/Qwen-7B-Chat-beta`, etc.)
|
- Qwen2 (`Qwen/Qwen1.5-7B`, `Qwen/Qwen1.5-7B-Chat`, etc.)
|
||||||
|
- Qwen2MoE (`Qwen/Qwen1.5-MoE-A2.7B`, `Qwen/Qwen1.5-MoE-A2.7B-Chat`, etc.)
|
||||||
- StableLM(`stabilityai/stablelm-3b-4e1t`, `stabilityai/stablelm-base-alpha-7b-v2`, etc.)
|
- StableLM(`stabilityai/stablelm-3b-4e1t`, `stabilityai/stablelm-base-alpha-7b-v2`, etc.)
|
||||||
|
- Starcoder2(`bigcode/starcoder2-3b`, `bigcode/starcoder2-7b`, `bigcode/starcoder2-15b`, etc.)
|
||||||
|
- Xverse (`xverse/XVERSE-7B-Chat`, `xverse/XVERSE-13B-Chat`, `xverse/XVERSE-65B-Chat`, etc.)
|
||||||
- Yi (`01-ai/Yi-6B`, `01-ai/Yi-34B`, etc.)
|
- Yi (`01-ai/Yi-6B`, `01-ai/Yi-34B`, etc.)
|
||||||
|
|
||||||
# Usage: OpenAI Compatibility
|
# Usage: OpenAI Compatibility
|
||||||
|
|||||||
@@ -1,51 +0,0 @@
|
|||||||
import os
|
|
||||||
import shutil
|
|
||||||
from huggingface_hub import snapshot_download
|
|
||||||
from vllm.model_executor.weight_utils import prepare_hf_model_weights, Disabledtqdm
|
|
||||||
|
|
||||||
def download_extras_or_tokenizer(model_name, cache_dir, revision, extras=False):
|
|
||||||
"""Download model or tokenizer and prepare its weights, returning the local folder path."""
|
|
||||||
pattern = ["*token*", "*.json"] if extras else None
|
|
||||||
extra_dir = "/extras" if extras else ""
|
|
||||||
folder = snapshot_download(
|
|
||||||
model_name,
|
|
||||||
cache_dir=cache_dir + extra_dir,
|
|
||||||
revision=revision,
|
|
||||||
tqdm_class=Disabledtqdm,
|
|
||||||
allow_patterns=pattern if extras else None,
|
|
||||||
ignore_patterns=["*.safetensors", "*.bin", "*.pt"] if not extras else None
|
|
||||||
)
|
|
||||||
return folder
|
|
||||||
|
|
||||||
def move_files(src_dir, dest_dir):
|
|
||||||
"""Move files from source to destination directory."""
|
|
||||||
for f in os.listdir(src_dir):
|
|
||||||
src_path = os.path.join(src_dir, f)
|
|
||||||
dst_path = os.path.join(dest_dir, f)
|
|
||||||
shutil.copy2(src_path, dst_path)
|
|
||||||
os.remove(src_path)
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
model, download_dir = os.getenv("MODEL_NAME"), os.getenv("HF_HOME")
|
|
||||||
tokenizer = os.getenv("TOKENIZER_NAME") or model
|
|
||||||
|
|
||||||
revisions = {
|
|
||||||
"model": os.getenv("MODEL_REVISION") or None,
|
|
||||||
"tokenizer": os.getenv("TOKENIZER_REVISION") or None
|
|
||||||
}
|
|
||||||
|
|
||||||
if not model or not download_dir:
|
|
||||||
raise ValueError(f"Must specify model and download_dir. Model: {model}, download_dir: {download_dir}")
|
|
||||||
|
|
||||||
os.makedirs(download_dir, exist_ok=True)
|
|
||||||
model_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights(model_name_or_path=model, revision=revisions["model"], cache_dir=download_dir)
|
|
||||||
model_extras_folder = download_extras_or_tokenizer(model, download_dir, revisions["model"], extras=True)
|
|
||||||
move_files(model_extras_folder, model_folder)
|
|
||||||
|
|
||||||
with open("/local_model_path.txt", "w") as f:
|
|
||||||
f.write(model_folder)
|
|
||||||
|
|
||||||
if tokenizer != model:
|
|
||||||
tokenizer_folder = download_extras_or_tokenizer(tokenizer, download_dir, revisions["tokenizer"])
|
|
||||||
with open("/local_tokenizer_path.txt", "w") as f:
|
|
||||||
f.write(tokenizer_folder)
|
|
||||||
@@ -1,4 +1,3 @@
|
|||||||
hf_transfer
|
|
||||||
ray
|
ray
|
||||||
pandas
|
pandas
|
||||||
pyarrow
|
pyarrow
|
||||||
@@ -6,4 +5,6 @@ runpod==1.6.2
|
|||||||
huggingface-hub
|
huggingface-hub
|
||||||
packaging
|
packaging
|
||||||
typing-extensions==4.7.1
|
typing-extensions==4.7.1
|
||||||
pydantic
|
pydantic
|
||||||
|
pydantic-settings
|
||||||
|
hf-transfer
|
||||||
@@ -0,0 +1,65 @@
|
|||||||
|
variable "PUSH" {
|
||||||
|
default = "true"
|
||||||
|
}
|
||||||
|
|
||||||
|
variable "REPOSITORY" {
|
||||||
|
default = "runpod"
|
||||||
|
}
|
||||||
|
|
||||||
|
variable "BASE_IMAGE_VERSION" {
|
||||||
|
default = "1.0.0"
|
||||||
|
}
|
||||||
|
|
||||||
|
group "all" {
|
||||||
|
targets = ["base", "main"]
|
||||||
|
}
|
||||||
|
|
||||||
|
group "base" {
|
||||||
|
targets = ["base-1180", "base-1210"]
|
||||||
|
}
|
||||||
|
|
||||||
|
group "main" {
|
||||||
|
targets = ["worker-1180", "worker-1210"]
|
||||||
|
}
|
||||||
|
|
||||||
|
target "base-1180" {
|
||||||
|
tags = ["${REPOSITORY}/worker-vllm:base-${BASE_IMAGE_VERSION}-cuda11.8.0"]
|
||||||
|
context = "vllm-base-image"
|
||||||
|
dockerfile = "Dockerfile"
|
||||||
|
args = {
|
||||||
|
WORKER_CUDA_VERSION = "11.8.0"
|
||||||
|
}
|
||||||
|
output = ["type=docker,push=${PUSH}"]
|
||||||
|
}
|
||||||
|
|
||||||
|
target "base-1210" {
|
||||||
|
tags = ["${REPOSITORY}/worker-vllm:base-${BASE_IMAGE_VERSION}-cuda12.1.0"]
|
||||||
|
context = "vllm-base-image"
|
||||||
|
dockerfile = "Dockerfile"
|
||||||
|
args = {
|
||||||
|
WORKER_CUDA_VERSION = "12.1.0"
|
||||||
|
}
|
||||||
|
output = ["type=docker,push=${PUSH}"]
|
||||||
|
}
|
||||||
|
|
||||||
|
target "worker-1180" {
|
||||||
|
tags = ["${REPOSITORY}/worker-vllm:${BASE_IMAGE_VERSION}-cuda11.8.0"]
|
||||||
|
context = "."
|
||||||
|
dockerfile = "Dockerfile"
|
||||||
|
args = {
|
||||||
|
BASE_IMAGE_VERSION = "${BASE_IMAGE_VERSION}"
|
||||||
|
WORKER_CUDA_VERSION = "11.8.0"
|
||||||
|
}
|
||||||
|
output = ["type=docker,push=${PUSH}"]
|
||||||
|
}
|
||||||
|
|
||||||
|
target "worker-1210" {
|
||||||
|
tags = ["${REPOSITORY}/worker-vllm:${BASE_IMAGE_VERSION}-cuda12.1.0"]
|
||||||
|
context = "."
|
||||||
|
dockerfile = "Dockerfile"
|
||||||
|
args = {
|
||||||
|
BASE_IMAGE_VERSION = "${BASE_IMAGE_VERSION}"
|
||||||
|
WORKER_CUDA_VERSION = "12.1.0"
|
||||||
|
}
|
||||||
|
output = ["type=docker,push=${PUSH}"]
|
||||||
|
}
|
||||||
Binary file not shown.
|
After Width: | Height: | Size: 27 MiB |
+30
-24
@@ -1,27 +1,31 @@
|
|||||||
import os
|
import os
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
from utils import count_physical_cores
|
|
||||||
from torch.cuda import device_count
|
from torch.cuda import device_count
|
||||||
|
from utils import get_int_bool_env
|
||||||
|
|
||||||
class EngineConfig:
|
class EngineConfig:
|
||||||
def __init__(self):
|
def __init__(self):
|
||||||
load_dotenv()
|
load_dotenv()
|
||||||
self.model_name_or_path, self.hf_home, self.model_revision = self._get_local_or_env("/local_model_path.txt", "MODEL_NAME")
|
self.hf_home = os.getenv("HF_HOME")
|
||||||
self.tokenizer_name_or_path, _, self.tokenizer_revision = self._get_local_or_env("/local_tokenizer_path.txt", "TOKENIZER_NAME")
|
# Check if /local_metadata.json exists
|
||||||
self.tokenizer_name_or_path = self.tokenizer_name_or_path or self.model_name_or_path
|
local_metadata = {}
|
||||||
self.quantization = self._get_quantization()
|
if os.path.exists("/local_metadata.json"):
|
||||||
|
with open("/local_metadata.json", "r") as f:
|
||||||
|
local_metadata = json.load(f)
|
||||||
|
if local_metadata.get("model_name") is None:
|
||||||
|
raise ValueError("Model name is not found in /local_metadata.json, there was a problem when you baked the model in.")
|
||||||
|
logging.info("Using baked-in model")
|
||||||
|
os.environ["TRANSFORMERS_OFFLINE"] = "1"
|
||||||
|
os.environ["HF_HUB_OFFLINE"] = "1"
|
||||||
|
|
||||||
|
self.model_name_or_path = local_metadata.get("model_name", os.getenv("MODEL_NAME"))
|
||||||
|
self.model_revision = local_metadata.get("revision", os.getenv("MODEL_REVISION"))
|
||||||
|
self.tokenizer_name_or_path = local_metadata.get("tokenizer_name", os.getenv("TOKENIZER_NAME")) or self.model_name_or_path
|
||||||
|
self.tokenizer_revision = local_metadata.get("tokenizer_revision", os.getenv("TOKENIZER_REVISION"))
|
||||||
|
self.quantization = local_metadata.get("quantization", os.getenv("QUANTIZATION"))
|
||||||
self.config = self._initialize_config()
|
self.config = self._initialize_config()
|
||||||
|
|
||||||
def _get_local_or_env(self, local_path, env_var):
|
|
||||||
if os.path.exists(local_path):
|
|
||||||
with open(local_path, "r") as file:
|
|
||||||
return file.read().strip(), None, None
|
|
||||||
return os.getenv(env_var), os.getenv("HF_HOME"), os.getenv(f"{env_var}_REVISION")
|
|
||||||
|
|
||||||
def _get_quantization(self):
|
|
||||||
quantization = os.getenv("QUANTIZATION", "").lower()
|
|
||||||
return quantization if quantization in ["awq", "squeezellm", "gptq"] else None
|
|
||||||
|
|
||||||
def _initialize_config(self):
|
def _initialize_config(self):
|
||||||
args = {
|
args = {
|
||||||
"model": self.model_name_or_path,
|
"model": self.model_name_or_path,
|
||||||
@@ -32,20 +36,22 @@ class EngineConfig:
|
|||||||
"dtype": os.getenv("DTYPE", "half" if self.quantization else "auto"),
|
"dtype": os.getenv("DTYPE", "half" if self.quantization else "auto"),
|
||||||
"tokenizer": self.tokenizer_name_or_path,
|
"tokenizer": self.tokenizer_name_or_path,
|
||||||
"tokenizer_revision": self.tokenizer_revision,
|
"tokenizer_revision": self.tokenizer_revision,
|
||||||
"disable_log_stats": bool(int(os.getenv("DISABLE_LOG_STATS", 1))),
|
"disable_log_stats": get_int_bool_env("DISABLE_LOG_STATS", True),
|
||||||
"disable_log_requests": bool(int(os.getenv("DISABLE_LOG_REQUESTS", 1))),
|
"disable_log_requests": get_int_bool_env("DISABLE_LOG_REQUESTS", True),
|
||||||
"trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))),
|
"trust_remote_code": get_int_bool_env("TRUST_REMOTE_CODE", False),
|
||||||
"gpu_memory_utilization": float(os.getenv("GPU_MEMORY_UTILIZATION", 0.95)),
|
"gpu_memory_utilization": float(os.getenv("GPU_MEMORY_UTILIZATION", 0.95)),
|
||||||
"max_parallel_loading_workers": None if device_count() > 1 or not os.getenv("MAX_PARALLEL_LOADING_WORKERS") else int(os.getenv("MAX_PARALLEL_LOADING_WORKERS")),
|
"max_parallel_loading_workers": None if device_count() > 1 or not os.getenv("MAX_PARALLEL_LOADING_WORKERS") else int(os.getenv("MAX_PARALLEL_LOADING_WORKERS")),
|
||||||
"max_model_len": int(os.getenv("MAX_MODEL_LENGTH")) if os.getenv("MAX_MODEL_LENGTH") else None,
|
"max_model_len": int(os.getenv("MAX_MODEL_LEN")) if os.getenv("MAX_MODEL_LEN") else None,
|
||||||
"tensor_parallel_size": device_count(),
|
"tensor_parallel_size": device_count(),
|
||||||
"seed": int(os.getenv("SEED")) if os.getenv("SEED") else None,
|
"seed": int(os.getenv("SEED")) if os.getenv("SEED") else None,
|
||||||
"kv_cache_dtype": os.getenv("KV_CACHE_DTYPE"),
|
"kv_cache_dtype": os.getenv("KV_CACHE_DTYPE"),
|
||||||
"block_size": int(os.getenv("BLOCK_SIZE")) if os.getenv("BLOCK_SIZE") else None,
|
"block_size": int(os.getenv("BLOCK_SIZE")) if os.getenv("BLOCK_SIZE") else None,
|
||||||
"swap_space": int(os.getenv("SWAP_SPACE")) if os.getenv("SWAP_SPACE") else None,
|
"swap_space": int(os.getenv("SWAP_SPACE")) if os.getenv("SWAP_SPACE") else None,
|
||||||
"max_context_len_to_capture": int(os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE")) if os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE") else None,
|
"max_context_len_to_capture": int(os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE")) if os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE") else None,
|
||||||
"disable_custom_all_reduce": bool(int(os.getenv("DISABLE_CUSTOM_ALL_REDUCE", 0))),
|
"disable_custom_all_reduce": get_int_bool_env("DISABLE_CUSTOM_ALL_REDUCE", False),
|
||||||
"enforce_eager": bool(int(os.getenv("ENFORCE_EAGER", 0)))
|
"enforce_eager": get_int_bool_env("ENFORCE_EAGER", False)
|
||||||
}
|
}
|
||||||
|
if args["kv_cache_dtype"] == "fp8_e5m2":
|
||||||
return {k: v for k, v in args.items() if v is not None}
|
args["kv_cache_dtype"] = "fp8"
|
||||||
|
logging.warning("Using fp8_e5m2 is deprecated. Please use fp8 instead.")
|
||||||
|
return {k: v for k, v in args.items() if v not in [None, ""]}
|
||||||
|
|||||||
+1
-27
@@ -1,30 +1,4 @@
|
|||||||
from typing import Union
|
|
||||||
|
|
||||||
DEFAULT_BATCH_SIZE = 50
|
DEFAULT_BATCH_SIZE = 50
|
||||||
DEFAULT_MAX_CONCURRENCY = 300
|
DEFAULT_MAX_CONCURRENCY = 300
|
||||||
DEFAULT_BATCH_SIZE_GROWTH_FACTOR = 3
|
DEFAULT_BATCH_SIZE_GROWTH_FACTOR = 3
|
||||||
DEFAULT_MIN_BATCH_SIZE = 1
|
DEFAULT_MIN_BATCH_SIZE = 1
|
||||||
|
|
||||||
SAMPLING_PARAM_TYPES = {
|
|
||||||
"n": int,
|
|
||||||
"best_of": int,
|
|
||||||
"presence_penalty": float,
|
|
||||||
"frequency_penalty": float,
|
|
||||||
"repetition_penalty": float,
|
|
||||||
"temperature": Union[float, int],
|
|
||||||
"top_p": float,
|
|
||||||
"top_k": int,
|
|
||||||
"min_p": float,
|
|
||||||
"use_beam_search": bool,
|
|
||||||
"length_penalty": float,
|
|
||||||
"early_stopping": Union[bool, str],
|
|
||||||
"stop": Union[str, list],
|
|
||||||
"stop_token_ids": list,
|
|
||||||
"ignore_eos": bool,
|
|
||||||
"max_tokens": int,
|
|
||||||
"logprobs": int,
|
|
||||||
"prompt_logprobs": int,
|
|
||||||
"skip_special_tokens": bool,
|
|
||||||
"spaces_between_special_tokens": bool,
|
|
||||||
"include_stop_str_in_output": bool
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
import os
|
||||||
|
from huggingface_hub import snapshot_download
|
||||||
|
import json
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
model_name = os.getenv("MODEL_NAME")
|
||||||
|
if not model_name:
|
||||||
|
raise ValueError("Must specify model name by adding --build-arg MODEL_NAME=<your model's repo>")
|
||||||
|
revision = os.getenv("MODEL_REVISION") or None
|
||||||
|
snapshot_download(model_name, revision=revision, cache_dir=os.getenv("HF_HOME"))
|
||||||
|
|
||||||
|
tokenizer_name = os.getenv("TOKENIZER_NAME") or None
|
||||||
|
tokenizer_revision = os.getenv("TOKENIZER_REVISION") or None
|
||||||
|
if tokenizer_name:
|
||||||
|
snapshot_download(tokenizer_name, revision=tokenizer_revision, cache_dir=os.getenv("HF_HOME"))
|
||||||
|
|
||||||
|
# Create file with metadata of baked in model and/or tokenizer
|
||||||
|
|
||||||
|
with open("/local_metadata.json", "w") as f:
|
||||||
|
json.dump({
|
||||||
|
"model_name": model_name,
|
||||||
|
"revision": revision,
|
||||||
|
"tokenizer_name": tokenizer_name or model_name,
|
||||||
|
"tokenizer_revision": tokenizer_revision or revision,
|
||||||
|
"quantization": os.getenv("QUANTIZATION")
|
||||||
|
}, f)
|
||||||
|
|
||||||
+12
-6
@@ -5,8 +5,9 @@ import json
|
|||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
from torch.cuda import device_count
|
from torch.cuda import device_count
|
||||||
from typing import AsyncGenerator
|
from typing import AsyncGenerator
|
||||||
|
import time
|
||||||
|
|
||||||
from vllm import AsyncLLMEngine, AsyncEngineArgs, SamplingParams
|
from vllm import AsyncLLMEngine, AsyncEngineArgs
|
||||||
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
|
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
|
||||||
from vllm.entrypoints.openai.serving_completion import OpenAIServingCompletion
|
from vllm.entrypoints.openai.serving_completion import OpenAIServingCompletion
|
||||||
from vllm.entrypoints.openai.protocol import ChatCompletionRequest, CompletionRequest, ErrorResponse
|
from vllm.entrypoints.openai.protocol import ChatCompletionRequest, CompletionRequest, ErrorResponse
|
||||||
@@ -16,7 +17,6 @@ from constants import DEFAULT_MAX_CONCURRENCY, DEFAULT_BATCH_SIZE, DEFAULT_BATCH
|
|||||||
from tokenizer import TokenizerWrapper
|
from tokenizer import TokenizerWrapper
|
||||||
from config import EngineConfig
|
from config import EngineConfig
|
||||||
|
|
||||||
|
|
||||||
class vLLMEngine:
|
class vLLMEngine:
|
||||||
def __init__(self, engine = None):
|
def __init__(self, engine = None):
|
||||||
load_dotenv() # For local development
|
load_dotenv() # For local development
|
||||||
@@ -35,7 +35,7 @@ class vLLMEngine:
|
|||||||
try:
|
try:
|
||||||
async for batch in self._generate_vllm(
|
async for batch in self._generate_vllm(
|
||||||
llm_input=job_input.llm_input,
|
llm_input=job_input.llm_input,
|
||||||
validated_sampling_params=job_input.validated_sampling_params,
|
validated_sampling_params=job_input.sampling_params,
|
||||||
batch_size=job_input.max_batch_size,
|
batch_size=job_input.max_batch_size,
|
||||||
stream=job_input.stream,
|
stream=job_input.stream,
|
||||||
apply_chat_template=job_input.apply_chat_template,
|
apply_chat_template=job_input.apply_chat_template,
|
||||||
@@ -45,12 +45,11 @@ class vLLMEngine:
|
|||||||
):
|
):
|
||||||
yield batch
|
yield batch
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
yield create_error_response(str(e)).model_dump()
|
yield {"error": create_error_response(str(e)).model_dump()}
|
||||||
|
|
||||||
async def _generate_vllm(self, llm_input, validated_sampling_params, batch_size, stream, apply_chat_template, request_id, batch_size_growth_factor, min_batch_size: str) -> AsyncGenerator[dict, None]:
|
async def _generate_vllm(self, llm_input, validated_sampling_params, batch_size, stream, apply_chat_template, request_id, batch_size_growth_factor, min_batch_size: str) -> AsyncGenerator[dict, None]:
|
||||||
if apply_chat_template or isinstance(llm_input, list):
|
if apply_chat_template or isinstance(llm_input, list):
|
||||||
llm_input = self.tokenizer.apply_chat_template(llm_input)
|
llm_input = self.tokenizer.apply_chat_template(llm_input)
|
||||||
validated_sampling_params = SamplingParams(**validated_sampling_params)
|
|
||||||
results_generator = self.llm.generate(llm_input, validated_sampling_params, request_id)
|
results_generator = self.llm.generate(llm_input, validated_sampling_params, request_id)
|
||||||
n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True
|
n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True
|
||||||
last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0}
|
last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0}
|
||||||
@@ -102,7 +101,11 @@ class vLLMEngine:
|
|||||||
|
|
||||||
def _initialize_llm(self):
|
def _initialize_llm(self):
|
||||||
try:
|
try:
|
||||||
return AsyncLLMEngine.from_engine_args(AsyncEngineArgs(**self.config))
|
start = time.time()
|
||||||
|
engine = AsyncLLMEngine.from_engine_args(AsyncEngineArgs(**self.config))
|
||||||
|
end = time.time()
|
||||||
|
logging.info(f"Initialized vLLM engine in {end - start:.2f}s")
|
||||||
|
return engine
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logging.error("Error initializing vLLM engine: %s", e)
|
logging.error("Error initializing vLLM engine: %s", e)
|
||||||
raise e
|
raise e
|
||||||
@@ -138,6 +141,9 @@ class OpenAIvLLMEngine:
|
|||||||
|
|
||||||
async def _handle_model_request(self):
|
async def _handle_model_request(self):
|
||||||
models = await self.chat_engine.show_available_models()
|
models = await self.chat_engine.show_available_models()
|
||||||
|
fixed_model = models.data[0]
|
||||||
|
fixed_model.id = self.served_model_name
|
||||||
|
models.data = [fixed_model]
|
||||||
return models.model_dump()
|
return models.model_dump()
|
||||||
|
|
||||||
async def _handle_chat_or_completion_request(self, openai_request: JobInput):
|
async def _handle_chat_or_completion_request(self, openai_request: JobInput):
|
||||||
|
|||||||
+11
-19
@@ -1,10 +1,9 @@
|
|||||||
|
import os
|
||||||
import logging
|
import logging
|
||||||
from http import HTTPStatus
|
from http import HTTPStatus
|
||||||
from typing import Any, Dict
|
|
||||||
from constants import SAMPLING_PARAM_TYPES
|
|
||||||
from vllm.utils import random_uuid
|
from vllm.utils import random_uuid
|
||||||
from vllm.entrypoints.openai.protocol import ErrorResponse
|
from vllm.entrypoints.openai.protocol import ErrorResponse
|
||||||
|
from vllm import SamplingParams
|
||||||
|
|
||||||
logging.basicConfig(level=logging.INFO)
|
logging.basicConfig(level=logging.INFO)
|
||||||
|
|
||||||
@@ -25,20 +24,6 @@ def count_physical_cores():
|
|||||||
|
|
||||||
return len(cores)
|
return len(cores)
|
||||||
|
|
||||||
def validate_sampling_params(params: Dict[str, Any]) -> Dict[str, Any]:
|
|
||||||
validated_params = {}
|
|
||||||
invalid_params = []
|
|
||||||
for key, value in params.items():
|
|
||||||
expected_type = SAMPLING_PARAM_TYPES.get(key)
|
|
||||||
if expected_type and isinstance(value, expected_type):
|
|
||||||
validated_params[key] = value
|
|
||||||
else:
|
|
||||||
invalid_params.append(key)
|
|
||||||
|
|
||||||
if len(invalid_params) > 0:
|
|
||||||
logging.warning("Ignoring invalid sampling params: %s", invalid_params)
|
|
||||||
|
|
||||||
return validated_params
|
|
||||||
|
|
||||||
class JobInput:
|
class JobInput:
|
||||||
def __init__(self, job):
|
def __init__(self, job):
|
||||||
@@ -47,7 +32,7 @@ class JobInput:
|
|||||||
self.max_batch_size = job.get("max_batch_size")
|
self.max_batch_size = job.get("max_batch_size")
|
||||||
self.apply_chat_template = job.get("apply_chat_template", False)
|
self.apply_chat_template = job.get("apply_chat_template", False)
|
||||||
self.use_openai_format = job.get("use_openai_format", False)
|
self.use_openai_format = job.get("use_openai_format", False)
|
||||||
self.validated_sampling_params = validate_sampling_params(job.get("sampling_params", {}))
|
self.sampling_params = SamplingParams(**job.get("sampling_params", {}))
|
||||||
self.request_id = random_uuid()
|
self.request_id = random_uuid()
|
||||||
batch_size_growth_factor = job.get("batch_size_growth_factor")
|
batch_size_growth_factor = job.get("batch_size_growth_factor")
|
||||||
self.batch_size_growth_factor = float(batch_size_growth_factor) if batch_size_growth_factor else None
|
self.batch_size_growth_factor = float(batch_size_growth_factor) if batch_size_growth_factor else None
|
||||||
@@ -78,4 +63,11 @@ class BatchSize:
|
|||||||
def create_error_response(message: str, err_type: str = "BadRequestError", status_code: HTTPStatus = HTTPStatus.BAD_REQUEST) -> ErrorResponse:
|
def create_error_response(message: str, err_type: str = "BadRequestError", status_code: HTTPStatus = HTTPStatus.BAD_REQUEST) -> ErrorResponse:
|
||||||
return ErrorResponse(message=message,
|
return ErrorResponse(message=message,
|
||||||
type=err_type,
|
type=err_type,
|
||||||
code=status_code.value)
|
code=status_code.value)
|
||||||
|
|
||||||
|
def get_int_bool_env(env_var: str, default: bool) -> bool:
|
||||||
|
return int(os.getenv(env_var, int(default))) == 1
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,149 @@
|
|||||||
|
################### vLLM Base Dockerfile ###################
|
||||||
|
# This Dockerfile is for building the image that the
|
||||||
|
# vLLM worker container will use as its base image.
|
||||||
|
# If your changes are outside of the vLLM source code, you
|
||||||
|
# do not need to build this image.
|
||||||
|
##########################################################
|
||||||
|
|
||||||
|
# Define the CUDA version for the build
|
||||||
|
ARG WORKER_CUDA_VERSION=11.8.0
|
||||||
|
|
||||||
|
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-devel-ubuntu22.04 AS dev
|
||||||
|
|
||||||
|
# Re-declare ARG after FROM
|
||||||
|
ARG WORKER_CUDA_VERSION
|
||||||
|
|
||||||
|
# Update and install dependencies
|
||||||
|
RUN apt-get update -y \
|
||||||
|
&& apt-get install -y python3-pip git
|
||||||
|
|
||||||
|
# Set working directory
|
||||||
|
WORKDIR /vllm-installation
|
||||||
|
|
||||||
|
RUN ldconfig /usr/local/cuda-$(echo "$WORKER_CUDA_VERSION" | sed 's/\.0$//')/compat/
|
||||||
|
|
||||||
|
# Install build and runtime dependencies
|
||||||
|
COPY vllm/requirements-common.txt requirements-common.txt
|
||||||
|
COPY vllm/requirements-cuda${WORKER_CUDA_VERSION}.txt requirements-cuda.txt
|
||||||
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
|
pip install -r requirements-cuda.txt
|
||||||
|
|
||||||
|
# Install development dependencies
|
||||||
|
COPY vllm/requirements-dev.txt requirements-dev.txt
|
||||||
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
|
pip install -r requirements-dev.txt
|
||||||
|
|
||||||
|
ARG torch_cuda_arch_list='7.0 7.5 8.0 8.6 8.9 9.0+PTX'
|
||||||
|
ENV TORCH_CUDA_ARCH_LIST=${torch_cuda_arch_list}
|
||||||
|
|
||||||
|
FROM dev AS build
|
||||||
|
|
||||||
|
# Re-declare ARG after FROM
|
||||||
|
ARG WORKER_CUDA_VERSION
|
||||||
|
|
||||||
|
# Install build dependencies
|
||||||
|
COPY vllm/requirements-build.txt requirements-build.txt
|
||||||
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
|
pip install -r requirements-build.txt
|
||||||
|
|
||||||
|
# install compiler cache to speed up compilation leveraging local or remote caching
|
||||||
|
RUN apt-get update -y && apt-get install -y ccache
|
||||||
|
|
||||||
|
# Copy necessary files
|
||||||
|
COPY vllm/csrc csrc
|
||||||
|
COPY vllm/setup.py setup.py
|
||||||
|
COPY vllm/cmake cmake
|
||||||
|
COPY vllm/CMakeLists.txt CMakeLists.txt
|
||||||
|
COPY vllm/requirements-common.txt requirements-common.txt
|
||||||
|
COPY vllm/requirements-cuda${WORKER_CUDA_VERSION}.txt requirements-cuda.txt
|
||||||
|
COPY vllm/pyproject.toml pyproject.toml
|
||||||
|
COPY vllm/vllm vllm
|
||||||
|
|
||||||
|
# Set environment variables for building extensions
|
||||||
|
ENV WORKER_CUDA_VERSION=${WORKER_CUDA_VERSION}
|
||||||
|
ENV VLLM_INSTALL_PUNICA_KERNELS=0
|
||||||
|
# Build extensions
|
||||||
|
ENV CCACHE_DIR=/root/.cache/ccache
|
||||||
|
RUN --mount=type=cache,target=/root/.cache/ccache \
|
||||||
|
--mount=type=cache,target=/root/.cache/pip \
|
||||||
|
python3 setup.py bdist_wheel --dist-dir=dist
|
||||||
|
|
||||||
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
|
pip cache remove vllm_nccl*
|
||||||
|
|
||||||
|
FROM dev as flash-attn-builder
|
||||||
|
# max jobs used for build
|
||||||
|
# flash attention version
|
||||||
|
ARG flash_attn_version=v2.5.8
|
||||||
|
ENV FLASH_ATTN_VERSION=${flash_attn_version}
|
||||||
|
|
||||||
|
WORKDIR /usr/src/flash-attention-v2
|
||||||
|
|
||||||
|
# Download the wheel or build it if a pre-compiled release doesn't exist
|
||||||
|
RUN pip --verbose wheel flash-attn==${FLASH_ATTN_VERSION} \
|
||||||
|
--no-build-isolation --no-deps --no-cache-dir
|
||||||
|
|
||||||
|
FROM dev as NCCL-installer
|
||||||
|
|
||||||
|
# Re-declare ARG after FROM
|
||||||
|
ARG WORKER_CUDA_VERSION
|
||||||
|
|
||||||
|
# Update and install necessary libraries
|
||||||
|
RUN apt-get update -y \
|
||||||
|
&& apt-get install -y wget
|
||||||
|
|
||||||
|
# Install NCCL library
|
||||||
|
RUN if [ "$WORKER_CUDA_VERSION" = "11.8.0" ]; then \
|
||||||
|
wget https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.0-1_all.deb \
|
||||||
|
&& dpkg -i cuda-keyring_1.0-1_all.deb \
|
||||||
|
&& apt-get update \
|
||||||
|
&& apt install -y libnccl2=2.15.5-1+cuda11.8 libnccl-dev=2.15.5-1+cuda11.8; \
|
||||||
|
elif [ "$WORKER_CUDA_VERSION" = "12.1.0" ]; then \
|
||||||
|
wget https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.0-1_all.deb \
|
||||||
|
&& dpkg -i cuda-keyring_1.0-1_all.deb \
|
||||||
|
&& apt-get update \
|
||||||
|
&& apt install -y libnccl2=2.17.1-1+cuda12.1 libnccl-dev=2.17.1-1+cuda12.1; \
|
||||||
|
else \
|
||||||
|
echo "Unsupported CUDA version: $WORKER_CUDA_VERSION"; \
|
||||||
|
exit 1; \
|
||||||
|
fi
|
||||||
|
|
||||||
|
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-base-ubuntu22.04 AS vllm-base
|
||||||
|
|
||||||
|
# Re-declare ARG after FROM
|
||||||
|
ARG WORKER_CUDA_VERSION
|
||||||
|
|
||||||
|
# Update and install necessary libraries
|
||||||
|
RUN apt-get update -y \
|
||||||
|
&& apt-get install -y python3-pip
|
||||||
|
|
||||||
|
# Set working directory
|
||||||
|
WORKDIR /vllm-workspace
|
||||||
|
|
||||||
|
RUN ldconfig /usr/local/cuda-$(echo "$WORKER_CUDA_VERSION" | sed 's/\.0$//')/compat/
|
||||||
|
|
||||||
|
RUN --mount=type=bind,from=build,src=/vllm-installation/dist,target=/vllm-workspace/dist \
|
||||||
|
--mount=type=cache,target=/root/.cache/pip \
|
||||||
|
pip install dist/*.whl --verbose
|
||||||
|
|
||||||
|
RUN --mount=type=bind,from=flash-attn-builder,src=/usr/src/flash-attention-v2,target=/usr/src/flash-attention-v2 \
|
||||||
|
--mount=type=cache,target=/root/.cache/pip \
|
||||||
|
pip install /usr/src/flash-attention-v2/*.whl --no-cache-dir
|
||||||
|
|
||||||
|
FROM vllm-base AS runtime
|
||||||
|
|
||||||
|
# install additional dependencies for openai api server
|
||||||
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
|
pip install accelerate hf_transfer modelscope tensorizer
|
||||||
|
|
||||||
|
# Set PYTHONPATH environment variable
|
||||||
|
ENV PYTHONPATH="/"
|
||||||
|
|
||||||
|
# Copy NCCL library
|
||||||
|
COPY --from=NCCL-installer /usr/lib/x86_64-linux-gnu/libnccl.so.2 /usr/lib/x86_64-linux-gnu/libnccl.so.2
|
||||||
|
# Set the VLLM_NCCL_SO_PATH environment variable
|
||||||
|
ENV VLLM_NCCL_SO_PATH="/usr/lib/x86_64-linux-gnu/libnccl.so.2"
|
||||||
|
|
||||||
|
|
||||||
|
# Validate the installation
|
||||||
|
RUN python3 -c "import vllm; print(vllm.__file__)"
|
||||||
Submodule
+1
Submodule vllm-base-image/vllm added at ba8f5e79e1
@@ -0,0 +1,2 @@
|
|||||||
|
version: '0.4.2'
|
||||||
|
dev_version: '0.4.2'
|
||||||
@@ -1,109 +0,0 @@
|
|||||||
################### vLLM Base Dockerfile ###################
|
|
||||||
# This Dockerfile is for building the image that the
|
|
||||||
# vLLM worker container will use as its base image.
|
|
||||||
# If your changes are outside of the vLLM source code, you
|
|
||||||
# do not need to build this image.
|
|
||||||
##########################################################
|
|
||||||
|
|
||||||
# Define the CUDA version for the build
|
|
||||||
ARG WORKER_CUDA_VERSION=11.8.0
|
|
||||||
|
|
||||||
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-devel-ubuntu22.04 AS dev
|
|
||||||
|
|
||||||
# Re-declare ARG after FROM
|
|
||||||
ARG WORKER_CUDA_VERSION
|
|
||||||
|
|
||||||
# Update and install dependencies
|
|
||||||
RUN apt-get update -y \
|
|
||||||
&& apt-get install -y python3-pip git
|
|
||||||
|
|
||||||
RUN if [ "${WORKER_CUDA_VERSION}" = "12.1.0" ]; then \
|
|
||||||
ldconfig /usr/local/cuda-12.1/compat/; \
|
|
||||||
fi
|
|
||||||
|
|
||||||
# Set working directory
|
|
||||||
WORKDIR /vllm-installation
|
|
||||||
|
|
||||||
# Install build and runtime dependencies
|
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/requirements.txt requirements.txt
|
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
|
||||||
pip install -r requirements.txt
|
|
||||||
|
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
|
||||||
if [ "${WORKER_CUDA_VERSION}" = "11.8.0" ]; then \
|
|
||||||
pip install -U --force-reinstall torch==2.1.2 xformers==0.0.23.post1 --index-url https://download.pytorch.org/whl/cu118; \
|
|
||||||
fi
|
|
||||||
|
|
||||||
# Install development dependencies
|
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/requirements-dev.txt requirements-dev.txt
|
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
|
||||||
pip install -r requirements-dev.txt
|
|
||||||
|
|
||||||
FROM dev AS build
|
|
||||||
|
|
||||||
# Re-declare ARG after FROM
|
|
||||||
ARG WORKER_CUDA_VERSION
|
|
||||||
|
|
||||||
# Install build dependencies
|
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/requirements-build.txt requirements-build.txt
|
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
|
||||||
pip install -r requirements-build.txt
|
|
||||||
|
|
||||||
# Copy necessary files
|
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/csrc csrc
|
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/setup.py setup.py
|
|
||||||
COPY vllm-12.1.0/pyproject.toml pyproject.toml
|
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/vllm/__init__.py vllm/__init__.py
|
|
||||||
|
|
||||||
# Conditional installation based on CUDA version
|
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
|
||||||
if [ "${WORKER_CUDA_VERSION}" = "11.8.0" ]; then \
|
|
||||||
pip install -U --force-reinstall torch==2.1.2 xformers==0.0.23.post1 --index-url https://download.pytorch.org/whl/cu118; \
|
|
||||||
rm pyproject.toml; \
|
|
||||||
elif [ "${WORKER_CUDA_VERSION}" != "12.1.0" ]; then \
|
|
||||||
echo "WORKER_CUDA_VERSION not supported"; \
|
|
||||||
exit 1; \
|
|
||||||
fi
|
|
||||||
|
|
||||||
# Set environment variables for building extensions
|
|
||||||
ARG torch_cuda_arch_list='7.0 7.5 8.0 8.6 8.9 9.0+PTX'
|
|
||||||
ENV TORCH_CUDA_ARCH_LIST=${torch_cuda_arch_list}
|
|
||||||
ARG max_jobs=48
|
|
||||||
ENV MAX_JOBS=${max_jobs}
|
|
||||||
ARG nvcc_threads=1024
|
|
||||||
ENV NVCC_THREADS=${nvcc_threads}
|
|
||||||
|
|
||||||
# Build extensions
|
|
||||||
RUN python3 setup.py build_ext --inplace
|
|
||||||
|
|
||||||
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-runtime-ubuntu22.04 AS vllm-base
|
|
||||||
|
|
||||||
# Re-declare ARG after FROM
|
|
||||||
ARG WORKER_CUDA_VERSION
|
|
||||||
|
|
||||||
# Update and install necessary libraries
|
|
||||||
RUN apt-get update -y \
|
|
||||||
&& apt-get install -y python3-pip
|
|
||||||
|
|
||||||
# Set working directory
|
|
||||||
WORKDIR /vllm-installation
|
|
||||||
|
|
||||||
# Install runtime dependencies
|
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/requirements.txt requirements.txt
|
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
|
||||||
pip install -r requirements.txt
|
|
||||||
|
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
|
||||||
if [ "${WORKER_CUDA_VERSION}" = "11.8.0" ]; then \
|
|
||||||
pip install -U --force-reinstall torch==2.1.2 xformers==0.0.23.post1 --index-url https://download.pytorch.org/whl/cu118; \
|
|
||||||
fi
|
|
||||||
|
|
||||||
# Copy built files from the build stage
|
|
||||||
COPY --from=build /vllm-installation/vllm/*.so /vllm-installation/vllm/
|
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/vllm vllm
|
|
||||||
|
|
||||||
# Set PYTHONPATH environment variable
|
|
||||||
ENV PYTHONPATH="/"
|
|
||||||
|
|
||||||
# Validate the installation
|
|
||||||
RUN python3 -c "import sys; print(sys.path); import vllm; print(vllm.__file__)"
|
|
||||||
@@ -1,12 +0,0 @@
|
|||||||
#!/bin/bash
|
|
||||||
|
|
||||||
git clone https://github.com/runpod/vllm-fork-for-sls-worker.git
|
|
||||||
|
|
||||||
cp -r vllm-fork-for-sls-worker vllm-12.1.0
|
|
||||||
cp -r vllm-fork-for-sls-worker vllm-11.8.0
|
|
||||||
rm -rf vllm-fork-for-sls-worker
|
|
||||||
|
|
||||||
cd vllm-11.8.0
|
|
||||||
git checkout cuda-11.8
|
|
||||||
|
|
||||||
echo "vLLM Base Image Builder Setup Complete."
|
|
||||||
Reference in New Issue
Block a user