Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2941db0fb8 | ||
|
|
bfeb60c54e | ||
|
|
7b3fd05542 | ||
|
|
b0e7b575f3 | ||
|
|
4f5e0d37c4 | ||
|
|
2b5b8dfb61 | ||
|
|
8de10468dd | ||
|
|
b1720a154d | ||
|
|
f4d7c75504 | ||
|
|
3adc9e3336 | ||
|
|
fd00a1ece3 | ||
|
|
664dd35782 | ||
|
|
370698442c | ||
|
|
3e2cd080a2 | ||
|
|
e7b340d73b | ||
|
|
46eee12819 | ||
|
|
12d6f0778e | ||
|
|
fa5556434c | ||
|
|
97726372c0 | ||
|
|
9fc8e1e54c | ||
|
|
4cebe66b36 | ||
|
|
368c5f87fb | ||
|
|
15cc7cd36f | ||
|
|
ef3c303743 | ||
|
|
584852f0f6 | ||
|
|
65454c024a | ||
|
|
f023e2e097 |
@@ -1,3 +0,0 @@
|
|||||||
MODEL_NAME="mistralai/Mistral-7B-Instruct-v0.1"
|
|
||||||
MODEL_BASE_PATH="./models"
|
|
||||||
DISABLE_LOG_STATS=0
|
|
||||||
+35
-36
@@ -1,49 +1,48 @@
|
|||||||
# Base image - Set default to CUDA 11.8
|
ARG WORKER_CUDA_VERSION=11.8.0
|
||||||
ARG WORKER_CUDA_VERSION=11.8
|
FROM runpod/worker-vllm:base-0.2.2-cuda${WORKER_CUDA_VERSION} AS vllm-base
|
||||||
FROM runpod/base:0.4.2-cuda${WORKER_CUDA_VERSION}.0 as builder
|
|
||||||
|
|
||||||
ARG WORKER_CUDA_VERSION=11.8 # Required duplicate to keep in scope
|
|
||||||
|
|
||||||
# Set Environment Variables
|
|
||||||
ENV WORKER_CUDA_VERSION=${WORKER_CUDA_VERSION} \
|
|
||||||
HF_DATASETS_CACHE="/runpod-volume/huggingface-cache/datasets" \
|
|
||||||
HUGGINGFACE_HUB_CACHE="/runpod-volume/huggingface-cache/hub" \
|
|
||||||
TRANSFORMERS_CACHE="/runpod-volume/huggingface-cache/hub" \
|
|
||||||
HF_TRANSFER=1
|
|
||||||
|
|
||||||
|
RUN apt-get update -y \
|
||||||
|
&& apt-get install -y python3-pip
|
||||||
|
|
||||||
# Install Python dependencies
|
# Install Python dependencies
|
||||||
COPY builder/requirements.txt /requirements.txt
|
COPY builder/requirements.txt /requirements.txt
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
python3.11 -m pip install --upgrade pip && \
|
python3 -m pip install --upgrade pip && \
|
||||||
python3.11 -m pip install --upgrade -r /requirements.txt && \
|
python3 -m pip install --upgrade -r /requirements.txt
|
||||||
rm /requirements.txt
|
|
||||||
|
|
||||||
# Install torch and vllm based on CUDA version
|
|
||||||
RUN if [[ "${WORKER_CUDA_VERSION}" == 11.8* ]]; then \
|
|
||||||
python3.11 -m pip install -U --force-reinstall torch==2.1.2 xformers==0.0.23.post1 --index-url https://download.pytorch.org/whl/cu118; \
|
|
||||||
python3.11 -m pip install -e git+https://github.com/runpod/vllm-fork-for-sls-worker.git@cuda-11.8#egg=vllm; \
|
|
||||||
else \
|
|
||||||
python3.11 -m pip install -e git+https://github.com/runpod/vllm-fork-for-sls-worker.git#egg=vllm; \
|
|
||||||
fi && \
|
|
||||||
rm -rf /root/.cache/pip
|
|
||||||
|
|
||||||
# Add source files
|
|
||||||
COPY src .
|
|
||||||
|
|
||||||
# Setup for Option 2: Building the Image with the Model included
|
# Setup for Option 2: Building the Image with the Model included
|
||||||
ARG MODEL_NAME=""
|
ARG MODEL_NAME=""
|
||||||
ARG MODEL_BASE_PATH="/runpod-volume/"
|
ARG TOKENIZER_NAME=""
|
||||||
ARG HF_TOKEN=""
|
ARG BASE_PATH="/runpod-volume"
|
||||||
ARG QUANTIZATION=""
|
ARG QUANTIZATION=""
|
||||||
RUN if [ -n "$MODEL_NAME" ]; then \
|
ARG MODEL_REVISION=""
|
||||||
export MODEL_BASE_PATH=$MODEL_BASE_PATH && \
|
ARG TOKENIZER_REVISION=""
|
||||||
export MODEL_NAME=$MODEL_NAME && \
|
|
||||||
python3.11 /download_model.py --model $MODEL_NAME; \
|
ENV MODEL_NAME=$MODEL_NAME \
|
||||||
|
MODEL_REVISION=$REVISION \
|
||||||
|
TOKENIZER_NAME=$TOKENIZER_NAME \
|
||||||
|
TOKENIZER_REVISION=$TOKENIZER_REVISION \
|
||||||
|
BASE_PATH=$BASE_PATH \
|
||||||
|
QUANTIZATION=$QUANTIZATION \
|
||||||
|
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
|
||||||
|
HUGGINGFACE_HUB_CACHE="${BASE_PATH}/huggingface-cache/hub" \
|
||||||
|
HF_HOME="${BASE_PATH}/huggingface-cache/hub" \
|
||||||
|
HF_TRANSFER=1
|
||||||
|
|
||||||
|
ENV PYTHONPATH="/:/vllm-installation"
|
||||||
|
|
||||||
|
COPY builder/download_model.py /download_model.py
|
||||||
|
RUN --mount=type=secret,id=HF_TOKEN,required=false \
|
||||||
|
if [ -f /run/secrets/HF_TOKEN ]; then \
|
||||||
|
export HF_TOKEN=$(cat /run/secrets/HF_TOKEN); \
|
||||||
fi && \
|
fi && \
|
||||||
if [ -n "$QUANTIZATION" ]; then \
|
if [ -n "$MODEL_NAME" ]; then \
|
||||||
export QUANTIZATION=$QUANTIZATION; \
|
python3 /download_model.py; \
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
# Add source files
|
||||||
|
COPY src /src
|
||||||
|
|
||||||
|
|
||||||
# Start the handler
|
# Start the handler
|
||||||
CMD ["python3.11", "/handler.py"]
|
CMD ["python3", "/src/handler.py"]
|
||||||
@@ -1,64 +1,147 @@
|
|||||||
<div align="center">
|
<div align="center">
|
||||||
|
|
||||||
<h1>vLLM 0.2.6 Endpoint | Serverless Worker </h1>
|
<h1> vLLM Serverless Endpoint Worker </h1>
|
||||||
|
|
||||||
[](https://github.com/runpod-workers/worker-vllm/actions/workflows/docker-build-release.yml)
|
[](https://github.com/runpod-workers/worker-vllm/actions/workflows/docker-build-release.yml)
|
||||||
|
|
||||||
🚀 | This serverless worker utilizes vLLM behind the scenes and is integrated into RunPod's serverless environment. It supports dynamic auto-scaling using the built-in RunPod autoscaling feature.
|
Deploy Blazing-fast LLMs powered by [vLLM](https://github.com/vllm-project/vllm) on RunPod Serverless in a few clicks.
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
|
### Worker vLLM 0.2.0 - What's New
|
||||||
|
- You no longer need a linux-based machine or NVIDIA GPUs to build the worker.
|
||||||
|
- Over 3x lighter Docker image size.
|
||||||
|
- OpenAI Chat Completion output format (optional to use).
|
||||||
|
- Extremely fast image build time.
|
||||||
|
- Docker Secrets-protected Hugging Face token support for building the image with a model baked in without exposing your token.
|
||||||
|
- Support for `n` and `best_of` sampling parameters, which allow you to generate multiple responses from a single prompt.
|
||||||
|
- New environment variables for various configuration.
|
||||||
|
- vLLM Version: 0.2.7
|
||||||
|
|
||||||
|
## Table of Contents
|
||||||
|
- [Setting up the Serverless Worker](#setting-up-the-serverless-worker)
|
||||||
|
- [Option 1: Deploy Any Model Using Pre-Built Docker Image [**RECOMMENDED**]](#option-1-deploy-any-model-using-pre-built-docker-image-recommended)
|
||||||
|
- [Prerequisites](#prerequisites)
|
||||||
|
- [Environment Variables](#environment-variables)
|
||||||
|
- [Option 2: Build Docker Image with Model Inside](#option-2-build-docker-image-with-model-inside)
|
||||||
|
- [Arguments](#arguments)
|
||||||
|
- [Example: Building an image with OpenChat-3.5](#example-building-an-image-with-openchat-35)
|
||||||
|
- [(Optional) Including Huggingface Token](#optional-including-huggingface-token)
|
||||||
|
- [Compatible Models](#compatible-models)
|
||||||
|
- [Usage](#usage)
|
||||||
|
- [Endpoint Model Inputs](#endpoint-model-inputs)
|
||||||
|
- [Text Input Formats](#text-input-formats)
|
||||||
|
- [1. `prompt`](#1-prompt)
|
||||||
|
- [2. `messages`](#2-messages)
|
||||||
|
- [Sampling Parameters](#sampling-parameters)
|
||||||
|
|
||||||
## Setting up the Serverless Worker
|
## Setting up the Serverless Worker
|
||||||
|
|
||||||
### Option 1: Deploy Any Model Using Pre-Built Docker Image
|
### Option 1: Deploy Any Model Using Pre-Built Docker Image [Recommended]
|
||||||
|
|
||||||
We now offer a pre-built Docker Image for the vLLM Worker that you can configure entirely with Environment Variables when creating the RunPod Serverless Endpoint:
|
We now offer a pre-built Docker Image for the vLLM Worker that you can configure entirely with Environment Variables when creating the RunPod Serverless Endpoint:
|
||||||
|
|
||||||
<div align="center">
|
<div align="center">
|
||||||
|
|
||||||
```runpod/worker-vllm:dev```
|
Stable Image: ```runpod/worker-vllm:0.2.3```
|
||||||
|
|
||||||
|
Development Image: ```runpod/worker-vllm:dev```
|
||||||
|
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
|
#### Prerequisites
|
||||||
|
- RunPod Account
|
||||||
|
|
||||||
#### Environment Variables
|
#### Environment Variables
|
||||||
- **Required**:
|
|
||||||
|
**Required**:
|
||||||
- `MODEL_NAME`: Hugging Face Model Repository (e.g., `openchat/openchat-3.5-1210`).
|
- `MODEL_NAME`: Hugging Face Model Repository (e.g., `openchat/openchat-3.5-1210`).
|
||||||
|
|
||||||
- **Optional**:
|
**Optional**:
|
||||||
|
- LLM Settings:
|
||||||
|
- `MODEL_REVISION`: Model revision to load (default: `None`).
|
||||||
- `MAX_MODEL_LENGTH`: Maximum number of tokens for the engine to be able to handle. (default: maximum supported by the model)
|
- `MAX_MODEL_LENGTH`: Maximum number of tokens for the engine to be able to handle. (default: maximum supported by the model)
|
||||||
- `MODEL_BASE_PATH`: Model storage directory (default: `/runpod-volume`).
|
- `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't)
|
||||||
|
- `LOAD_FORMAT`: Format to load model in (default: `auto`).
|
||||||
- `HF_TOKEN`: Hugging Face token for private and gated models (e.g., Llama, Falcon).
|
- `HF_TOKEN`: Hugging Face token for private and gated models (e.g., Llama, Falcon).
|
||||||
- `NUM_GPU_SHARD`: Number of GPUs to split the model across. (default: `1`)
|
|
||||||
- `QUANTIZATION`: AWQ (`awq`), SqueezeLLM (`squeezellm`) or GPTQ (`gptq`) Quantization. The specified Model Repo must be of a quantized model. (default: `None`)
|
- `QUANTIZATION`: AWQ (`awq`), SqueezeLLM (`squeezellm`) or GPTQ (`gptq`) Quantization. The specified Model Repo must be of a quantized model. (default: `None`)
|
||||||
- `TRUST_REMOTE_CODE`: Whether to trust remote code with Hugging Face. (default: `0`)
|
- `TRUST_REMOTE_CODE`: Trust remote code for Hugging Face (default: `0`)
|
||||||
|
|
||||||
|
- Tokenizer Settings:
|
||||||
|
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer)
|
||||||
|
- `TOKENIZER_REVISION`: Tokenizer revision to load (default: `None`).
|
||||||
|
- `CUSTOM_CHAT_TEMPLATE`: Custom chat jinja template, read more about Hugging Face chat templates [here](https://huggingface.co/docs/transformers/chat_templating). (default: `None`)
|
||||||
|
|
||||||
|
- Tensor Parallelism:
|
||||||
|
Note that the more GPUs you split a model's weights accross, the slower it will be due to inter-GPU communication overhead. If you can fit the model on a single GPU, it is recommended to do so.
|
||||||
|
- `TENSOR_PARALLEL_SIZE`: Number of GPUs to shard the model across (default: `1`).
|
||||||
|
- If you are having issues loading your model with Tensor Parallelism, try decreasing `VLLM_CPU_FRACTION` (default: `1`).
|
||||||
|
|
||||||
|
- System Settings:
|
||||||
|
- `GPU_MEMORY_UTILIZATION`: GPU VRAM utilization (default: `0.98`).
|
||||||
|
- `MAX_PARALLEL_LOADING_WORKERS`: Maximum number of parallel workers for loading models, for non-Tensor Parallel only. (default: `number of available CPU cores` if `TENSOR_PARALLEL_SIZE` is `1`, otherwise `None`).
|
||||||
|
|
||||||
|
|
||||||
|
- Serverless Settings:
|
||||||
- `MAX_CONCURRENCY`: Max concurrent requests. (default: `100`)
|
- `MAX_CONCURRENCY`: Max concurrent requests. (default: `100`)
|
||||||
- `DEFAULT_BATCH_SIZE`: Token streaming batch size (default: `30`). This reduces the number of HTTP calls, increasing speed 8-10x vs non-batching, matching non-streaming performance.
|
- `DEFAULT_BATCH_SIZE`: Token streaming batch size (default: `30`). This reduces the number of HTTP calls, increasing speed 8-10x vs non-batching, matching non-streaming performance.
|
||||||
|
- `ALLOW_OPENAI_FORMAT`: Whether to allow users to specify `use_openai_format` to get output in OpenAI format. (default: `1`)
|
||||||
- `DISABLE_LOG_STATS`: Enable (`0`) or disable (`1`) vLLM stats logging.
|
- `DISABLE_LOG_STATS`: Enable (`0`) or disable (`1`) vLLM stats logging.
|
||||||
- `DISABLE_LOG_REQUESTS`: Enable (`0`) or disable (`1`) request logging.
|
- `DISABLE_LOG_REQUESTS`: Enable (`0`) or disable (`1`) request logging.
|
||||||
|
|
||||||
### Option 2: Build Docker Image with Model Inside
|
### Option 2: Build Docker Image with Model Inside
|
||||||
[!WARNING] If you are getting errors while building the image, try adding `ENV MAX_JOBS` to the Dockerfile and increase Docker memory limit to at least 25GB.
|
To build an image with the model baked in, you must specify the following docker arguments when building the image.
|
||||||
|
|
||||||
To build an image with the model baked in, you must specify the following docker arguments when building the image:
|
#### Prerequisites
|
||||||
|
- RunPod Account
|
||||||
|
- Docker
|
||||||
|
|
||||||
#### Arguments:
|
#### Arguments:
|
||||||
- **Required**
|
- **Required**
|
||||||
- `MODEL_NAME`
|
- `MODEL_NAME`
|
||||||
- **Optional**
|
- **Optional**
|
||||||
- `MODEL_BASE_PATH`: Defaults to `/runpod-volume` for network storage. Use `/models` or for local container storage.
|
- `MODEL_REVISION`: Model revision to load (default: `main`).
|
||||||
|
- `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't. If your intention is to bake the model into the image, you should set this to something like `/models` to make sure there are no issues if you were to accidentally attach network storage.)
|
||||||
- `QUANTIZATION`
|
- `QUANTIZATION`
|
||||||
- `HF_TOKEN`
|
- `WORKER_CUDA_VERSION`: `11.8.0` or `12.1.0` (default: `11.8.0` due to a small amount of workers not having CUDA 12.1 support yet. `12.1.0` is recommended for optimal performance).
|
||||||
- `WORKER_CUDA_VERSION`: `11.8` or `12.1` (default: `11.8` due to a small amount of workers not having CUDA 12.1 support yet. `12.1` is recommended for optimal performance).
|
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer)
|
||||||
|
- `TOKENIZER_REVISION`: Tokenizer revision to load (default: `main`).
|
||||||
|
|
||||||
|
For the remaining settings, you may apply them as environment variables when running the container. Supported environment variables are listed in the [Environment Variables](#environment-variables) section.
|
||||||
|
|
||||||
#### Example: Building an image with OpenChat-3.5
|
#### Example: Building an image with OpenChat-3.5
|
||||||
`sudo docker build -t username/image:tag --build-arg MODEL_NAME="openchat/openchat_3.5" --build-arg MODEL_BASE_PATH="/models" .`
|
```bash
|
||||||
|
sudo docker build -t username/image:tag --build-arg MODEL_NAME="openchat/openchat_3.5" --build-arg BASE_PATH="/models" .
|
||||||
|
```
|
||||||
|
|
||||||
### Compatible Models
|
##### (Optional) Including Huggingface Token
|
||||||
- LLaMA & LLaMA-2 (`meta-llama/Llama-2-70b-hf`, `lmsys/vicuna-13b-v1.3`, `young-geng/koala`, `openlm-research/open_llama_13b`, etc.)
|
If the model you would like to deploy is private or gated, you will need to include it during build time as a Docker secret, which will protect it from being exposed in the image and on DockerHub.
|
||||||
|
1. Enable Docker BuildKit (required for secrets).
|
||||||
|
```bash
|
||||||
|
export DOCKER_BUILDKIT=1
|
||||||
|
```
|
||||||
|
2. Export your Hugging Face token as an environment variable
|
||||||
|
```bash
|
||||||
|
export HF_TOKEN="your_token_here"
|
||||||
|
```
|
||||||
|
2. Add the token as a secret when building
|
||||||
|
```bash
|
||||||
|
docker build -t username/image:tag --secret id=HF_TOKEN --build-arg MODEL_NAME="openchat/openchat_3.5" .
|
||||||
|
```
|
||||||
|
|
||||||
|
### Compatible Model Architectures
|
||||||
- Mistral (`mistralai/Mistral-7B-v0.1`, `mistralai/Mistral-7B-Instruct-v0.1`, etc.)
|
- Mistral (`mistralai/Mistral-7B-v0.1`, `mistralai/Mistral-7B-Instruct-v0.1`, etc.)
|
||||||
- Mixtral (`mistralai/Mixtral-8x7B-v0.1`, `mistralai/Mixtral-8x7B-Instruct-v0.1`, etc.)
|
- Mixtral (`mistralai/Mixtral-8x7B-v0.1`, `mistralai/Mixtral-8x7B-Instruct-v0.1`, etc.)
|
||||||
|
- Phi (`microsoft/phi-1_5`, `microsoft/phi-2`, etc.)
|
||||||
|
- LLaMA & LLaMA-2 (`meta-llama/Llama-2-70b-hf`, `lmsys/vicuna-13b-v1.3`, `young-geng/koala`, `openlm-research/open_llama_13b`, etc.)
|
||||||
|
- Qwen2 (`Qwen/Qwen2-7B-beta`, `Qwen/Qwen-7B-Chat-beta`, etc.)
|
||||||
|
- StableLM(`stabilityai/stablelm-3b-4e1t`, `stabilityai/stablelm-base-alpha-7b-v2`, etc.)
|
||||||
|
- Yi (`01-ai/Yi-6B`, `01-ai/Yi-34B`, etc.)
|
||||||
|
- Qwen (`Qwen/Qwen-7B`, `Qwen/Qwen-7B-Chat`, etc.)
|
||||||
- Aquila & Aquila2 (`BAAI/AquilaChat2-7B`, `BAAI/AquilaChat2-34B`, `BAAI/Aquila-7B`, `BAAI/AquilaChat-7B`, etc.)
|
- Aquila & Aquila2 (`BAAI/AquilaChat2-7B`, `BAAI/AquilaChat2-34B`, `BAAI/Aquila-7B`, `BAAI/AquilaChat-7B`, etc.)
|
||||||
- Baichuan & Baichuan2 (`baichuan-inc/Baichuan2-13B-Chat`, `baichuan-inc/Baichuan-7B`, etc.)
|
- Baichuan & Baichuan2 (`baichuan-inc/Baichuan2-13B-Chat`, `baichuan-inc/Baichuan-7B`, etc.)
|
||||||
- BLOOM (`bigscience/bloom`, `bigscience/bloomz`, etc.)
|
- BLOOM (`bigscience/bloom`, `bigscience/bloomz`, etc.)
|
||||||
- ChatGLM (`THUDM/chatglm2-6b`, `THUDM/chatglm3-6b`, etc.)
|
- ChatGLM (`THUDM/chatglm2-6b`, `THUDM/chatglm3-6b`, etc.)
|
||||||
|
- DeciLM (`Deci/DeciLM-7B`, `Deci/DeciLM-7B-instruct`, etc.)
|
||||||
- Falcon (`tiiuae/falcon-7b`, `tiiuae/falcon-40b`, `tiiuae/falcon-rw-7b`, etc.)
|
- Falcon (`tiiuae/falcon-7b`, `tiiuae/falcon-40b`, `tiiuae/falcon-rw-7b`, etc.)
|
||||||
- GPT-2 (`gpt2`, `gpt2-xl`, etc.)
|
- GPT-2 (`gpt2`, `gpt2-xl`, etc.)
|
||||||
- GPT BigCode (`bigcode/starcoder`, `bigcode/gpt_bigcode-santacoder`, etc.)
|
- GPT BigCode (`bigcode/starcoder`, `bigcode/gpt_bigcode-santacoder`, etc.)
|
||||||
@@ -67,57 +150,61 @@ To build an image with the model baked in, you must specify the following docker
|
|||||||
- InternLM (`internlm/internlm-7b`, `internlm/internlm-chat-7b`, etc.)
|
- InternLM (`internlm/internlm-7b`, `internlm/internlm-chat-7b`, etc.)
|
||||||
- MPT (`mosaicml/mpt-7b`, `mosaicml/mpt-30b`, etc.)
|
- MPT (`mosaicml/mpt-7b`, `mosaicml/mpt-30b`, etc.)
|
||||||
- OPT (`facebook/opt-66b`, `facebook/opt-iml-max-30b`, etc.)
|
- OPT (`facebook/opt-66b`, `facebook/opt-iml-max-30b`, etc.)
|
||||||
- Phi (`microsoft/phi-1_5`, `microsoft/phi-2`, etc.)
|
|
||||||
- Qwen (`Qwen/Qwen-7B`, `Qwen/Qwen-7B-Chat`, etc.)
|
|
||||||
- Yi (`01-ai/Yi-6B`, `01-ai/Yi-34B`, etc.)
|
|
||||||
|
|
||||||
And any other models supported by vLLM 0.2.6.
|
|
||||||
|
|
||||||
|
|
||||||
Ensure that you have Docker installed and properly set up before running the docker build commands. Once built, you can deploy this serverless worker in your desired environment with confidence that it will automatically scale based on demand. For further inquiries or assistance, feel free to contact our support team.
|
## Usage
|
||||||
|
### Endpoint Model Inputs
|
||||||
|
|
||||||
## Model Inputs
|
|
||||||
You may either use a `prompt` or a list of `messages` as input. If you use `messages`, the model's chat template will be applied to the messages automatically, so the model must have one. If you use `prompt`, you may optionally apply the model's chat template to the prompt by setting `apply_chat_template` to `true`.
|
You may either use a `prompt` or a list of `messages` as input. If you use `messages`, the model's chat template will be applied to the messages automatically, so the model must have one. If you use `prompt`, you may optionally apply the model's chat template to the prompt by setting `apply_chat_template` to `true`.
|
||||||
| Argument | Type | Default | Description |
|
| Argument | Type | Default | Description |
|
||||||
|-----------------|------|--------------------|-----------------------------------------------------------------------------------------------|
|
|-----------------------|----------------------|--------------------|--------------------------------------------------------------------------------------------------------|
|
||||||
| `prompt` | str | | Prompt string to generate text based on. |
|
| `prompt` | str | | Prompt string to generate text based on. |
|
||||||
| `messages` | list[dict[str, str]] | | List of messages, which will automatically have the model's chat template applied. Overrides `prompt`. |
|
| `messages` | list[dict[str, str]] | | List of messages, which will automatically have the model's chat template applied. Overrides `prompt`. |
|
||||||
| `apply_chat_template` | bool | False | Whether to apply the model's chat template to the `prompt`. |
|
| `use_openai_format` | bool | False | Whether to return output in OpenAI format. `ALLOW_OPENAI_FORMAT` environment variable must be `1`, the input should preferably be a `messages` list, but `prompt` is accepted. |
|
||||||
| `sampling_params` | dict | {} | Sampling parameters to control the generation, like temperature, top_p, etc. |
|
| `apply_chat_template` | bool | False | Whether to apply the model's chat template to the `prompt`. |
|
||||||
| `stream` | bool | False | Whether to enable streaming of output. If True, responses are streamed as they are generated. |
|
| `sampling_params` | dict | {} | Sampling parameters to control the generation, like temperature, top_p, etc. |
|
||||||
| `batch_size` | int | DEFAULT_BATCH_SIZE | The number of tokens to stream every HTTP POST call. |
|
| `stream` | bool | False | Whether to enable streaming of output. If True, responses are streamed as they are generated. |
|
||||||
|
| `batch_size` | int | DEFAULT_BATCH_SIZE | The number of tokens to stream every HTTP POST call. |
|
||||||
|
|
||||||
### Messages Format
|
### Text Input Formats
|
||||||
|
You may either use a `prompt` or a list of `messages` as input.
|
||||||
|
#### 1. `prompt`
|
||||||
|
The prompt string can be any string, and the model's chat template will not be applied to it unless `apply_chat_template` is set to `true`, in which case it will be treated as a user message.
|
||||||
|
|
||||||
|
Example:
|
||||||
|
```json
|
||||||
|
"prompt": "..."
|
||||||
|
```
|
||||||
|
#### 2. `messages`
|
||||||
Your list can contain any number of messages, and each message can have any role from the following list:
|
Your list can contain any number of messages, and each message can have any role from the following list:
|
||||||
- `user`
|
- `user`
|
||||||
- `assistant`
|
- `assistant`
|
||||||
- `system`
|
- `system`
|
||||||
|
|
||||||
The model's chat template will be applied to the messages automatically.
|
The model's chat template will be applied to the messages automatically, so the model must have one.
|
||||||
|
|
||||||
Example:
|
Example:
|
||||||
```json
|
```json
|
||||||
[
|
"messages": [
|
||||||
{
|
{
|
||||||
"role": "system",
|
"role": "system",
|
||||||
"content": "..."
|
"content": "..."
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"role": "user",
|
"role": "user",
|
||||||
"content": "..."
|
"content": "..."
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"role": "assistant",
|
"role": "assistant",
|
||||||
"content": "..."
|
"content": "..."
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
```
|
```
|
||||||
|
|
||||||
### Sampling Parameters
|
### Sampling Parameters
|
||||||
| Argument | Type | Default | Description |
|
| Argument | Type | Default | Description |
|
||||||
|-------------------------------|-----------------------------|---------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
|---------------------------------|-----------------------------|---------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||||
| `best_of` | Optional[int] | None | Number of output sequences generated from the prompt. The top `n` sequences are returned from these `best_of` sequences. Must be ≥ `n`. Treated as beam width in beam search. Default is `n`. |
|
| `n` | int | 1 | Number of output sequences generated from the prompt. The top `n` sequences are returned. |
|
||||||
|
| `best_of` | Optional[int] | `n` | Number of output sequences generated from the prompt. The top `n` sequences are returned from these `best_of` sequences. Must be ≥ `n`. Treated as beam width in beam search. Default is `n`. |
|
||||||
| `presence_penalty` | float | 0.0 | Penalizes new tokens based on their presence in the generated text so far. Values > 0 encourage new tokens, values < 0 encourage repetition. |
|
| `presence_penalty` | float | 0.0 | Penalizes new tokens based on their presence in the generated text so far. Values > 0 encourage new tokens, values < 0 encourage repetition. |
|
||||||
| `frequency_penalty` | float | 0.0 | Penalizes new tokens based on their frequency in the generated text so far. Values > 0 encourage new tokens, values < 0 encourage repetition. |
|
| `frequency_penalty` | float | 0.0 | Penalizes new tokens based on their frequency in the generated text so far. Values > 0 encourage new tokens, values < 0 encourage repetition. |
|
||||||
| `repetition_penalty` | float | 1.0 | Penalizes new tokens based on their appearance in the prompt and generated text. Values > 1 encourage new tokens, values < 1 encourage repetition. |
|
| `repetition_penalty` | float | 1.0 | Penalizes new tokens based on their appearance in the prompt and generated text. Values > 1 encourage new tokens, values < 1 encourage repetition. |
|
||||||
@@ -134,4 +221,3 @@ Example:
|
|||||||
| `max_tokens` | int | 16 | Maximum number of tokens to generate per output sequence. |
|
| `max_tokens` | int | 16 | Maximum number of tokens to generate per output sequence. |
|
||||||
| `skip_special_tokens` | bool | True | Whether to skip special tokens in the output. |
|
| `skip_special_tokens` | bool | True | Whether to skip special tokens in the output. |
|
||||||
| `spaces_between_special_tokens` | bool | True | Whether to add spaces between special tokens in the output. |
|
| `spaces_between_special_tokens` | bool | True | Whether to add spaces between special tokens in the output. |
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,51 @@
|
|||||||
|
import os
|
||||||
|
import shutil
|
||||||
|
from huggingface_hub import snapshot_download
|
||||||
|
from vllm.model_executor.weight_utils import prepare_hf_model_weights, Disabledtqdm
|
||||||
|
|
||||||
|
def download_extras_or_tokenizer(model_name, cache_dir, revision, extras=False):
|
||||||
|
"""Download model or tokenizer and prepare its weights, returning the local folder path."""
|
||||||
|
pattern = ["*token*", "*.json"] if extras else None
|
||||||
|
extra_dir = "/extras" if extras else ""
|
||||||
|
folder = snapshot_download(
|
||||||
|
model_name,
|
||||||
|
cache_dir=cache_dir + extra_dir,
|
||||||
|
revision=revision,
|
||||||
|
tqdm_class=Disabledtqdm,
|
||||||
|
allow_patterns=pattern if extras else None,
|
||||||
|
ignore_patterns=["*.safetensors", "*.bin", "*.pt"] if not extras else None
|
||||||
|
)
|
||||||
|
return folder
|
||||||
|
|
||||||
|
def move_files(src_dir, dest_dir):
|
||||||
|
"""Move files from source to destination directory."""
|
||||||
|
for f in os.listdir(src_dir):
|
||||||
|
src_path = os.path.join(src_dir, f)
|
||||||
|
dst_path = os.path.join(dest_dir, f)
|
||||||
|
shutil.copy2(src_path, dst_path)
|
||||||
|
os.remove(src_path)
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
model, download_dir = os.getenv("MODEL_NAME"), os.getenv("HF_HOME")
|
||||||
|
tokenizer = os.getenv("TOKENIZER_NAME") or model
|
||||||
|
|
||||||
|
revisions = {
|
||||||
|
"model": os.getenv("MODEL_REVISION") or None,
|
||||||
|
"tokenizer": os.getenv("TOKENIZER_REVISION") or None
|
||||||
|
}
|
||||||
|
|
||||||
|
if not model or not download_dir:
|
||||||
|
raise ValueError(f"Must specify model and download_dir. Model: {model}, download_dir: {download_dir}")
|
||||||
|
|
||||||
|
os.makedirs(download_dir, exist_ok=True)
|
||||||
|
model_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights(model_name_or_path=model, revision=revisions["model"], cache_dir=download_dir)
|
||||||
|
model_extras_folder = download_extras_or_tokenizer(model, download_dir, revisions["model"], extras=True)
|
||||||
|
move_files(model_extras_folder, model_folder)
|
||||||
|
|
||||||
|
with open("/local_model_path.txt", "w") as f:
|
||||||
|
f.write(model_folder)
|
||||||
|
|
||||||
|
if tokenizer != model:
|
||||||
|
tokenizer_folder = download_extras_or_tokenizer(tokenizer, download_dir, revisions["tokenizer"])
|
||||||
|
with open("/local_tokenizer_path.txt", "w") as f:
|
||||||
|
f.write(tokenizer_folder)
|
||||||
@@ -1,6 +1,9 @@
|
|||||||
hf_transfer
|
hf_transfer
|
||||||
runpod==1.5.2
|
ray
|
||||||
|
pandas
|
||||||
|
pyarrow
|
||||||
|
runpod==1.5.3
|
||||||
huggingface-hub
|
huggingface-hub
|
||||||
packaging
|
packaging
|
||||||
typing-extensions==4.7.1
|
typing-extensions==4.7.1
|
||||||
pydantic
|
pydantic
|
||||||
+7
-4
@@ -1,20 +1,22 @@
|
|||||||
|
from typing import Union
|
||||||
|
|
||||||
DEFAULT_BATCH_SIZE = 30
|
DEFAULT_BATCH_SIZE = 30
|
||||||
DEFAULT_MAX_CONCURRENCY = 300
|
DEFAULT_MAX_CONCURRENCY = 300
|
||||||
|
|
||||||
sampling_param_types = {
|
SAMPLING_PARAM_TYPES = {
|
||||||
"n": int,
|
"n": int,
|
||||||
"best_of": int,
|
"best_of": int,
|
||||||
"presence_penalty": float,
|
"presence_penalty": float,
|
||||||
"frequency_penalty": float,
|
"frequency_penalty": float,
|
||||||
"repetition_penalty": float,
|
"repetition_penalty": float,
|
||||||
"temperature": float,
|
"temperature": Union[float, int],
|
||||||
"top_p": float,
|
"top_p": float,
|
||||||
"top_k": int,
|
"top_k": int,
|
||||||
"min_p": float,
|
"min_p": float,
|
||||||
"use_beam_search": bool,
|
"use_beam_search": bool,
|
||||||
"length_penalty": float,
|
"length_penalty": float,
|
||||||
"early_stopping": (bool, str),
|
"early_stopping": Union[bool, str],
|
||||||
"stop": (str, list),
|
"stop": Union[str, list],
|
||||||
"stop_token_ids": list,
|
"stop_token_ids": list,
|
||||||
"ignore_eos": bool,
|
"ignore_eos": bool,
|
||||||
"max_tokens": int,
|
"max_tokens": int,
|
||||||
@@ -22,4 +24,5 @@ sampling_param_types = {
|
|||||||
"prompt_logprobs": int,
|
"prompt_logprobs": int,
|
||||||
"skip_special_tokens": bool,
|
"skip_special_tokens": bool,
|
||||||
"spaces_between_special_tokens": bool,
|
"spaces_between_special_tokens": bool,
|
||||||
|
"include_stop_str_in_output": bool
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,22 +0,0 @@
|
|||||||
import argparse
|
|
||||||
import os
|
|
||||||
from vllm.model_executor.weight_utils import prepare_hf_model_weights
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
parser = argparse.ArgumentParser()
|
|
||||||
parser.add_argument("--model", type=str)
|
|
||||||
parser.add_argument(
|
|
||||||
"--download_dir", type=str, default=os.environ.get("MODEL_BASE_PATH")
|
|
||||||
)
|
|
||||||
|
|
||||||
args = parser.parse_args()
|
|
||||||
if not args.model or not args.download_dir:
|
|
||||||
raise ValueError("Must specify model and download_dir")
|
|
||||||
|
|
||||||
if not os.path.exists(args.download_dir):
|
|
||||||
os.makedirs(args.download_dir)
|
|
||||||
|
|
||||||
prepare_hf_model_weights(
|
|
||||||
model_name_or_path=args.model,
|
|
||||||
cache_dir=args.download_dir,
|
|
||||||
)
|
|
||||||
+188
-34
@@ -1,17 +1,24 @@
|
|||||||
import os
|
import os
|
||||||
import logging
|
import logging
|
||||||
from typing import Union
|
from typing import Union, AsyncGenerator
|
||||||
import torch
|
import json
|
||||||
from vllm import AsyncLLMEngine, AsyncEngineArgs
|
from torch.cuda import device_count
|
||||||
|
from vllm import AsyncLLMEngine, AsyncEngineArgs, SamplingParams
|
||||||
|
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
|
||||||
|
from vllm.entrypoints.openai.protocol import ChatCompletionRequest
|
||||||
from transformers import AutoTokenizer
|
from transformers import AutoTokenizer
|
||||||
from utils import ServerlessConfig
|
from utils import count_physical_cores, DummyRequest
|
||||||
|
from constants import DEFAULT_MAX_CONCURRENCY
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
|
|
||||||
|
|
||||||
class Tokenizer:
|
class Tokenizer:
|
||||||
def __init__(self, model_name: str):
|
def __init__(self, tokenizer_name_or_path, tokenizer_revision, trust_remote_code):
|
||||||
self.tokenizer = AutoTokenizer.from_pretrained(model_name)
|
self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_name_or_path, revision=tokenizer_revision, trust_remote_code=trust_remote_code)
|
||||||
self.has_chat_template = bool(self.tokenizer.chat_template)
|
self.custom_chat_template = os.getenv("CUSTOM_CHAT_TEMPLATE")
|
||||||
|
self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template)
|
||||||
|
if self.custom_chat_template and isinstance(self.custom_chat_template, str):
|
||||||
|
self.tokenizer.chat_template = self.custom_chat_template
|
||||||
|
|
||||||
def apply_chat_template(self, input: Union[str, list[dict[str, str]]]) -> str:
|
def apply_chat_template(self, input: Union[str, list[dict[str, str]]]) -> str:
|
||||||
if isinstance(input, list):
|
if isinstance(input, list):
|
||||||
@@ -30,23 +37,152 @@ class Tokenizer:
|
|||||||
|
|
||||||
|
|
||||||
class vLLMEngine:
|
class vLLMEngine:
|
||||||
def __init__(self):
|
def __init__(self, engine = None):
|
||||||
load_dotenv() # For local development
|
load_dotenv() # For local development
|
||||||
self.config = self._initialize_config()
|
self.config = self._initialize_config()
|
||||||
self.serverless_config = ServerlessConfig()
|
logging.info("vLLM config: %s", self.config)
|
||||||
self.tokenizer = Tokenizer(self.config["model"])
|
self.tokenizer = Tokenizer(self.config["tokenizer"], self.config["tokenizer_revision"], self.config["trust_remote_code"])
|
||||||
self.llm = self._initialize_llm()
|
self.llm = self._initialize_llm() if engine is None else engine
|
||||||
|
self.openai_engine = self._initialize_openai()
|
||||||
|
self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
|
||||||
|
|
||||||
|
async def generate(self, job_input):
|
||||||
|
generator_args = job_input.__dict__
|
||||||
|
|
||||||
|
if generator_args.pop("use_openai_format"):
|
||||||
|
if self.openai_engine is None:
|
||||||
|
raise ValueError("OpenAI Chat Completion Format is not enabled for this model")
|
||||||
|
generator = self.generate_openai_chat
|
||||||
|
else:
|
||||||
|
generator = self.generate_vllm
|
||||||
|
|
||||||
|
async for batch in generator(**generator_args):
|
||||||
|
yield batch
|
||||||
|
|
||||||
|
async def generate_vllm(self, llm_input, validated_sampling_params, batch_size, stream, apply_chat_template, request_id: str) -> AsyncGenerator[dict, None]:
|
||||||
|
if apply_chat_template or isinstance(llm_input, list):
|
||||||
|
llm_input = self.tokenizer.apply_chat_template(llm_input)
|
||||||
|
validated_sampling_params = SamplingParams(**validated_sampling_params)
|
||||||
|
results_generator = self.llm.generate(llm_input, validated_sampling_params, request_id)
|
||||||
|
n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True
|
||||||
|
last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0}
|
||||||
|
|
||||||
|
batch = {
|
||||||
|
"choices": [{"tokens": []} for _ in range(n_responses)],
|
||||||
|
}
|
||||||
|
|
||||||
|
async for request_output in results_generator:
|
||||||
|
if is_first_output: # Count input tokens only once
|
||||||
|
n_input_tokens = len(request_output.prompt_token_ids)
|
||||||
|
is_first_output = False
|
||||||
|
|
||||||
|
for output in request_output.outputs:
|
||||||
|
output_index = output.index
|
||||||
|
token_counters["total"] += 1
|
||||||
|
if stream:
|
||||||
|
new_output = output.text[len(last_output_texts[output_index]):]
|
||||||
|
batch["choices"][output_index]["tokens"].append(new_output)
|
||||||
|
token_counters["batch"] += 1
|
||||||
|
|
||||||
|
if token_counters["batch"] >= batch_size:
|
||||||
|
batch["usage"] = {
|
||||||
|
"input": n_input_tokens,
|
||||||
|
"output": token_counters["total"],
|
||||||
|
}
|
||||||
|
yield batch
|
||||||
|
batch = {
|
||||||
|
"choices": [{"tokens": []} for _ in range(n_responses)],
|
||||||
|
}
|
||||||
|
token_counters["batch"] = 0
|
||||||
|
|
||||||
|
last_output_texts[output_index] = output.text
|
||||||
|
|
||||||
|
if not stream:
|
||||||
|
for output_index, output in enumerate(last_output_texts):
|
||||||
|
batch["choices"][output_index]["tokens"] = [output]
|
||||||
|
token_counters["batch"] += 1
|
||||||
|
|
||||||
|
if token_counters["batch"] > 0:
|
||||||
|
batch["usage"] = {"input": n_input_tokens, "output": token_counters["total"]}
|
||||||
|
yield batch
|
||||||
|
|
||||||
|
async def generate_openai_chat(self, llm_input, validated_sampling_params, batch_size, stream, apply_chat_template, request_id: str) -> AsyncGenerator[dict, None]:
|
||||||
|
|
||||||
|
if isinstance(llm_input, str):
|
||||||
|
llm_input = [{"role": "user", "content": llm_input}]
|
||||||
|
logging.warning("OpenAI Chat Completion format requires list input, converting to list and assigning 'user' role")
|
||||||
|
|
||||||
|
if not self.openai_engine:
|
||||||
|
raise ValueError("OpenAI Chat Completion format is disabled")
|
||||||
|
|
||||||
|
chat_completion_request = ChatCompletionRequest(
|
||||||
|
model=self.config["model"],
|
||||||
|
messages=llm_input,
|
||||||
|
stream=stream,
|
||||||
|
**validated_sampling_params,
|
||||||
|
)
|
||||||
|
|
||||||
|
response_generator = await self.openai_engine.create_chat_completion(chat_completion_request, DummyRequest())
|
||||||
|
if not stream:
|
||||||
|
yield json.loads(response_generator.model_dump_json())
|
||||||
|
else:
|
||||||
|
batch_contents = {}
|
||||||
|
batch_latest_choices = {}
|
||||||
|
batch_token_counter = 0
|
||||||
|
last_chunk = {}
|
||||||
|
|
||||||
|
async for chunk_str in response_generator:
|
||||||
|
try:
|
||||||
|
chunk = json.loads(chunk_str.removeprefix("data: ").rstrip("\n\n"))
|
||||||
|
except:
|
||||||
|
continue
|
||||||
|
|
||||||
|
if "choices" in chunk:
|
||||||
|
for choice in chunk["choices"]:
|
||||||
|
choice_index = choice["index"]
|
||||||
|
if "delta" in choice and "content" in choice["delta"]:
|
||||||
|
batch_contents[choice_index] = batch_contents.get(choice_index, []) + [choice["delta"]["content"]]
|
||||||
|
batch_latest_choices[choice_index] = choice
|
||||||
|
batch_token_counter += 1
|
||||||
|
last_chunk = chunk
|
||||||
|
|
||||||
|
if batch_token_counter >= batch_size:
|
||||||
|
for choice_index in batch_latest_choices:
|
||||||
|
batch_latest_choices[choice_index]["delta"]["content"] = batch_contents[choice_index]
|
||||||
|
last_chunk["choices"] = list(batch_latest_choices.values())
|
||||||
|
yield last_chunk
|
||||||
|
|
||||||
|
batch_contents = {}
|
||||||
|
batch_latest_choices = {}
|
||||||
|
batch_token_counter = 0
|
||||||
|
|
||||||
|
if batch_contents:
|
||||||
|
for choice_index in batch_latest_choices:
|
||||||
|
batch_latest_choices[choice_index]["delta"]["content"] = batch_contents[choice_index]
|
||||||
|
last_chunk["choices"] = list(batch_latest_choices.values())
|
||||||
|
yield last_chunk
|
||||||
|
|
||||||
def _initialize_config(self):
|
def _initialize_config(self):
|
||||||
|
quantization = self._get_quantization()
|
||||||
|
model, download_dir, model_revision = self._get_model_info()
|
||||||
|
tokenizer_name_or_path, tokenizer_revision = self._get_tokenizer_info()
|
||||||
|
if not tokenizer_name_or_path:
|
||||||
|
tokenizer_name_or_path = model
|
||||||
|
|
||||||
return {
|
return {
|
||||||
"model": os.getenv("MODEL_NAME"),
|
"model": model,
|
||||||
"download_dir": os.getenv("MODEL_BASE_PATH", "/runpod-volume/"),
|
"revision": model_revision,
|
||||||
"quantization": self._get_quantization(),
|
"download_dir": download_dir,
|
||||||
"dtype": "auto" if os.getenv("QUANTIZATION") is None else "half",
|
"quantization": quantization,
|
||||||
|
"load_format": os.getenv("LOAD_FORMAT", "auto"),
|
||||||
|
"dtype": "half" if quantization else "auto",
|
||||||
|
"tokenizer": tokenizer_name_or_path,
|
||||||
|
"tokenizer_revision": tokenizer_revision,
|
||||||
"disable_log_stats": bool(int(os.getenv("DISABLE_LOG_STATS", 1))),
|
"disable_log_stats": bool(int(os.getenv("DISABLE_LOG_STATS", 1))),
|
||||||
"disable_log_requests": bool(int(os.getenv("DISABLE_LOG_REQUESTS", 1))),
|
"disable_log_requests": bool(int(os.getenv("DISABLE_LOG_REQUESTS", 1))),
|
||||||
"trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))),
|
"trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))),
|
||||||
"gpu_memory_utilization": float(os.getenv("GPU_MEMORY_UTILIZATION", 0.98)),
|
"gpu_memory_utilization": float(os.getenv("GPU_MEMORY_UTILIZATION", 0.95)),
|
||||||
|
"max_parallel_loading_workers": self._get_max_parallel_loading_workers(),
|
||||||
"max_model_len": self._get_max_model_len(),
|
"max_model_len": self._get_max_model_len(),
|
||||||
"tensor_parallel_size": self._get_num_gpu_shard(),
|
"tensor_parallel_size": self._get_num_gpu_shard(),
|
||||||
}
|
}
|
||||||
@@ -57,18 +193,45 @@ class vLLMEngine:
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
logging.error("Error initializing vLLM engine: %s", e)
|
logging.error("Error initializing vLLM engine: %s", e)
|
||||||
raise e
|
raise e
|
||||||
|
|
||||||
|
def _initialize_openai(self):
|
||||||
|
if bool(int(os.getenv("ALLOW_OPENAI_FORMAT", 1))) and self.tokenizer.has_chat_template:
|
||||||
|
return OpenAIServingChat(self.llm, self.config["model"], "assistant", self.tokenizer.tokenizer.chat_template)
|
||||||
|
else:
|
||||||
|
return None
|
||||||
|
|
||||||
|
def _get_max_parallel_loading_workers(self):
|
||||||
|
if int(os.getenv("TENSOR_PARALLEL_SIZE", 1)) > 1:
|
||||||
|
return None
|
||||||
|
else:
|
||||||
|
return int(os.getenv("MAX_PARALLEL_LOADING_WORKERS", count_physical_cores()))
|
||||||
|
|
||||||
|
def _get_model_info(self):
|
||||||
|
if os.path.exists("/local_model_path.txt"):
|
||||||
|
model, download_dir, revision = open("/local_model_path.txt", "r").read().strip(), None, None
|
||||||
|
logging.info("Using local model at %s", model)
|
||||||
|
else:
|
||||||
|
model, download_dir, revision = os.getenv("MODEL_NAME"), os.getenv("HF_HOME"), os.getenv("MODEL_REVISION") or None
|
||||||
|
return model, download_dir, revision
|
||||||
|
|
||||||
|
def _get_tokenizer_info(self):
|
||||||
|
if os.path.exists("/local_tokenizer_path.txt"):
|
||||||
|
tokenizer_name_or_path, revision = open("/local_tokenizer_path.txt", "r").read().strip(), None
|
||||||
|
logging.info("Using local tokenizer at %s", tokenizer_name_or_path)
|
||||||
|
else:
|
||||||
|
tokenizer_name_or_path, revision = os.getenv("TOKENIZER_NAME"), os.getenv("TOKENIZER_REVISION") or None
|
||||||
|
return tokenizer_name_or_path, revision
|
||||||
|
|
||||||
def _get_num_gpu_shard(self):
|
def _get_num_gpu_shard(self):
|
||||||
final_num_gpu_shard = 1
|
num_gpu_shard = int(os.getenv("TENSOR_PARALLEL_SIZE", 1))
|
||||||
if bool(int(os.getenv("USE_TENSOR_PARALLEL", 0))):
|
if num_gpu_shard > 1:
|
||||||
env_num_gpu_shard = int(os.getenv("TENSOR_PARALLEL_SIZE", 1))
|
num_gpu_available = device_count()
|
||||||
num_gpu_available = torch.cuda.device_count()
|
num_gpu_shard = min(num_gpu_shard, num_gpu_available)
|
||||||
final_num_gpu_shard = min(env_num_gpu_shard, num_gpu_available)
|
logging.info("Using %s GPU shards", num_gpu_shard)
|
||||||
logging.info("Using %s GPU shards", final_num_gpu_shard)
|
return num_gpu_shard
|
||||||
return final_num_gpu_shard
|
|
||||||
|
|
||||||
def _get_max_model_len(self):
|
def _get_max_model_len(self):
|
||||||
max_model_len = os.getenv("MAX_MODEL_LEN")
|
max_model_len = os.getenv("MAX_MODEL_LENGTH")
|
||||||
return int(max_model_len) if max_model_len is not None else None
|
return int(max_model_len) if max_model_len is not None else None
|
||||||
|
|
||||||
def _get_n_current_jobs(self):
|
def _get_n_current_jobs(self):
|
||||||
@@ -77,13 +240,4 @@ class vLLMEngine:
|
|||||||
|
|
||||||
def _get_quantization(self):
|
def _get_quantization(self):
|
||||||
quantization = os.getenv("QUANTIZATION", "").lower()
|
quantization = os.getenv("QUANTIZATION", "").lower()
|
||||||
return quantization if quantization in ["awq", "squeezellm", "gptq"] else None
|
return quantization if quantization in ["awq", "squeezellm", "gptq"] else None
|
||||||
|
|
||||||
def concurrency_modifier(self, current_concurrency):
|
|
||||||
n_current_jobs = self._get_n_current_jobs()
|
|
||||||
requested_concurrency = max(0, self.serverless_config.max_concurrency - n_current_jobs)
|
|
||||||
if not self.config["disable_log_stats"]:
|
|
||||||
logging.info("Current Jobs: %s", n_current_jobs)
|
|
||||||
logging.info("Concurrency Modifier Requested Jobs: %s", requested_concurrency)
|
|
||||||
return requested_concurrency
|
|
||||||
|
|
||||||
+9
-57
@@ -1,67 +1,19 @@
|
|||||||
#!/usr/bin/env python
|
|
||||||
from typing import Generator
|
|
||||||
import runpod
|
import runpod
|
||||||
from utils import validate_sampling_params, random_uuid
|
from utils import JobInput
|
||||||
from engine import vLLMEngine
|
from engine import vLLMEngine
|
||||||
|
|
||||||
vllm_engine = vLLMEngine()
|
vllm_engine = vLLMEngine()
|
||||||
async def handler(job: dict) -> Generator[dict, None, None]:
|
|
||||||
job_input = job["input"]
|
|
||||||
llm_input = job_input.get("messages", job_input.get("prompt"))
|
|
||||||
apply_chat_template = job_input.get("apply_chat_template", False)
|
|
||||||
|
|
||||||
if apply_chat_template or isinstance(llm_input, list):
|
|
||||||
llm_input = vllm_engine.tokenizer.apply_chat_template(llm_input)
|
|
||||||
|
|
||||||
stream = job_input.get("stream", False)
|
|
||||||
batch_size = job_input.get("batch_size", vllm_engine.serverless_config.default_batch_size)
|
|
||||||
sampling_params = job_input.get("sampling_params", {})
|
|
||||||
|
|
||||||
validated_params = validate_sampling_params(sampling_params)
|
|
||||||
request_id = random_uuid()
|
|
||||||
results_generator = vllm_engine.llm.generate(
|
|
||||||
llm_input, validated_params, request_id
|
|
||||||
)
|
|
||||||
|
|
||||||
batch = {"tokens": []}
|
|
||||||
last_output_text = ""
|
|
||||||
n_input_tokens, is_first_output = 0, True
|
|
||||||
|
|
||||||
async for request_output in results_generator:
|
|
||||||
if is_first_output: # Count input tokens only once
|
|
||||||
n_input_tokens = len(request_output.prompt_token_ids)
|
|
||||||
is_first_output = False
|
|
||||||
|
|
||||||
for output in request_output.outputs:
|
|
||||||
if stream:
|
|
||||||
|
|
||||||
batch["tokens"].append(
|
|
||||||
output.text[len(last_output_text):]
|
|
||||||
)
|
|
||||||
finished = request_output.finished
|
|
||||||
if len(batch["tokens"]) >= batch_size or finished:
|
|
||||||
batch["usage"] = {
|
|
||||||
"input": n_input_tokens,
|
|
||||||
"output": len(output.token_ids),
|
|
||||||
}
|
|
||||||
batch["finished"] = finished
|
|
||||||
yield batch
|
|
||||||
batch = {"tokens": []}
|
|
||||||
|
|
||||||
last_output_text = output.text
|
|
||||||
|
|
||||||
if not stream:
|
|
||||||
yield {"tokens": [last_output_text],
|
|
||||||
"usage": {
|
|
||||||
"input": n_input_tokens,
|
|
||||||
"output": len(output.token_ids),
|
|
||||||
},
|
|
||||||
"finished": True}
|
|
||||||
|
|
||||||
|
async def handler(job):
|
||||||
|
job_input = JobInput(job["input"])
|
||||||
|
results_generator = vllm_engine.generate(job_input)
|
||||||
|
async for batch in results_generator:
|
||||||
|
yield batch
|
||||||
|
|
||||||
runpod.serverless.start(
|
runpod.serverless.start(
|
||||||
{
|
{
|
||||||
"handler": handler,
|
"handler": handler,
|
||||||
"concurrency_modifier": lambda x: vllm_engine.serverless_config.max_concurrency,
|
"concurrency_modifier": lambda x: vllm_engine.max_concurrency,
|
||||||
"return_aggregate_stream": True,
|
"return_aggregate_stream": True,
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
+38
-37
@@ -1,51 +1,52 @@
|
|||||||
import os
|
|
||||||
import logging
|
import logging
|
||||||
from typing import Any, Dict
|
from typing import Any, Dict
|
||||||
from vllm import SamplingParams
|
|
||||||
from vllm.utils import random_uuid
|
from vllm.utils import random_uuid
|
||||||
from constants import sampling_param_types, DEFAULT_BATCH_SIZE, DEFAULT_MAX_CONCURRENCY
|
from constants import SAMPLING_PARAM_TYPES, DEFAULT_BATCH_SIZE
|
||||||
|
|
||||||
logging.basicConfig(level=logging.INFO)
|
logging.basicConfig(level=logging.INFO)
|
||||||
|
|
||||||
|
def count_physical_cores():
|
||||||
|
with open('/proc/cpuinfo') as f:
|
||||||
|
content = f.readlines()
|
||||||
|
|
||||||
class ServerlessConfig:
|
cores = set()
|
||||||
def __init__(self):
|
current_physical_id = None
|
||||||
self._max_concurrency = int(
|
current_core_id = None
|
||||||
os.environ.get("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY)
|
|
||||||
)
|
|
||||||
self._default_batch_size = int(
|
|
||||||
os.environ.get("DEFAULT_BATCH_SIZE", DEFAULT_BATCH_SIZE)
|
|
||||||
)
|
|
||||||
|
|
||||||
@property
|
for line in content:
|
||||||
def max_concurrency(self):
|
if 'physical id' in line:
|
||||||
return self._max_concurrency
|
current_physical_id = line.strip().split(': ')[1]
|
||||||
|
elif 'core id' in line:
|
||||||
|
current_core_id = line.strip().split(': ')[1]
|
||||||
|
cores.add((current_physical_id, current_core_id))
|
||||||
|
|
||||||
@property
|
return len(cores)
|
||||||
def default_batch_size(self):
|
|
||||||
return self._default_batch_size
|
|
||||||
|
|
||||||
|
def validate_sampling_params(params: Dict[str, Any]) -> Dict[str, Any]:
|
||||||
def validate_sampling_params(params: Dict[str, Any]) -> SamplingParams:
|
|
||||||
validated_params = {}
|
validated_params = {}
|
||||||
|
invalid_params = []
|
||||||
for key, value in params.items():
|
for key, value in params.items():
|
||||||
expected_type = sampling_param_types.get(key)
|
expected_type = SAMPLING_PARAM_TYPES.get(key)
|
||||||
if value is None:
|
if expected_type and isinstance(value, expected_type):
|
||||||
validated_params[key] = None
|
validated_params[key] = value
|
||||||
continue
|
|
||||||
|
|
||||||
if expected_type is None:
|
|
||||||
continue
|
|
||||||
|
|
||||||
if isinstance(expected_type, tuple):
|
|
||||||
casted_value = next(
|
|
||||||
(t(value) for t in expected_type if isinstance(value, t)), None
|
|
||||||
)
|
|
||||||
else:
|
else:
|
||||||
casted_value = value if isinstance(value, expected_type) else None
|
invalid_params.append(key)
|
||||||
|
|
||||||
|
if len(invalid_params) > 0:
|
||||||
|
logging.warning("Ignoring invalid sampling params: %s", invalid_params)
|
||||||
|
|
||||||
|
return validated_params
|
||||||
|
|
||||||
if casted_value is not None:
|
class JobInput:
|
||||||
validated_params[key] = casted_value
|
def __init__(self, job):
|
||||||
|
self.llm_input = job.get("messages", job.get("prompt"))
|
||||||
return SamplingParams(**validated_params)
|
self.stream = job.get("stream", False)
|
||||||
|
self.batch_size = job.get("batch_size", DEFAULT_BATCH_SIZE)
|
||||||
|
self.apply_chat_template = job.get("apply_chat_template", False)
|
||||||
|
self.use_openai_format = job.get("use_openai_format", False)
|
||||||
|
self.validated_sampling_params = validate_sampling_params(job.get("sampling_params", {}))
|
||||||
|
self.request_id = random_uuid()
|
||||||
|
|
||||||
|
class DummyRequest:
|
||||||
|
async def is_disconnected(self):
|
||||||
|
return False
|
||||||
@@ -0,0 +1,100 @@
|
|||||||
|
################### vLLM Base Dockerfile ###################
|
||||||
|
# This Dockerfile is for building the image that the
|
||||||
|
# vLLM worker container will use as its base image.
|
||||||
|
# If your changes are outside of the vLLM source code, you
|
||||||
|
# do not need to build this image.
|
||||||
|
##########################################################
|
||||||
|
|
||||||
|
# Define the CUDA version for the build
|
||||||
|
ARG WORKER_CUDA_VERSION=12.1.0
|
||||||
|
|
||||||
|
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-devel-ubuntu22.04 AS dev
|
||||||
|
|
||||||
|
# Re-declare ARG after FROM
|
||||||
|
ARG WORKER_CUDA_VERSION
|
||||||
|
|
||||||
|
# Update and install dependencies
|
||||||
|
RUN apt-get update -y \
|
||||||
|
&& apt-get install -y python3-pip git
|
||||||
|
|
||||||
|
# Set working directory
|
||||||
|
WORKDIR /vllm-installation
|
||||||
|
|
||||||
|
# Install build and runtime dependencies
|
||||||
|
COPY vllm-${WORKER_CUDA_VERSION}/requirements.txt requirements.txt
|
||||||
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
|
pip install -r requirements.txt
|
||||||
|
|
||||||
|
# Install development dependencies
|
||||||
|
COPY vllm-${WORKER_CUDA_VERSION}/requirements-dev.txt requirements-dev.txt
|
||||||
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
|
pip install -r requirements-dev.txt
|
||||||
|
|
||||||
|
FROM dev AS build
|
||||||
|
|
||||||
|
# Re-declare ARG after FROM
|
||||||
|
ARG WORKER_CUDA_VERSION
|
||||||
|
|
||||||
|
# Install build dependencies
|
||||||
|
COPY vllm-${WORKER_CUDA_VERSION}/requirements-build.txt requirements-build.txt
|
||||||
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
|
pip install -r requirements-build.txt
|
||||||
|
|
||||||
|
# Copy necessary files
|
||||||
|
COPY vllm-${WORKER_CUDA_VERSION}/csrc csrc
|
||||||
|
COPY vllm-${WORKER_CUDA_VERSION}/setup.py setup.py
|
||||||
|
COPY vllm-12.1.0/pyproject.toml pyproject.toml
|
||||||
|
COPY vllm-${WORKER_CUDA_VERSION}/vllm/__init__.py vllm/__init__.py
|
||||||
|
|
||||||
|
# Conditional installation based on CUDA version
|
||||||
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
|
if [ "${WORKER_CUDA_VERSION}" = "11.8.0" ]; then \
|
||||||
|
pip install -U --force-reinstall torch==2.1.2 xformers==0.0.23.post1 --index-url https://download.pytorch.org/whl/cu118; \
|
||||||
|
rm pyproject.toml; \
|
||||||
|
elif [ "${WORKER_CUDA_VERSION}" != "12.1.0" ]; then \
|
||||||
|
echo "WORKER_CUDA_VERSION not supported"; \
|
||||||
|
exit 1; \
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Set environment variables for building extensions
|
||||||
|
ARG torch_cuda_arch_list='7.0 7.5 8.0 8.6 8.9 9.0+PTX'
|
||||||
|
ENV TORCH_CUDA_ARCH_LIST=${torch_cuda_arch_list}
|
||||||
|
ARG max_jobs=48
|
||||||
|
ENV MAX_JOBS=${max_jobs}
|
||||||
|
ARG nvcc_threads=1024
|
||||||
|
ENV NVCC_THREADS=${nvcc_threads}
|
||||||
|
|
||||||
|
# Build extensions
|
||||||
|
RUN python3 setup.py build_ext --inplace
|
||||||
|
|
||||||
|
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-base-ubuntu22.04 AS vllm-base
|
||||||
|
|
||||||
|
# Re-declare ARG after FROM
|
||||||
|
ARG WORKER_CUDA_VERSION
|
||||||
|
|
||||||
|
# Update and install necessary libraries
|
||||||
|
RUN apt-get update -y \
|
||||||
|
&& apt-get install -y python3-pip
|
||||||
|
|
||||||
|
# Set working directory
|
||||||
|
WORKDIR /vllm-installation
|
||||||
|
|
||||||
|
# Install runtime dependencies
|
||||||
|
COPY vllm-${WORKER_CUDA_VERSION}/requirements.txt requirements.txt
|
||||||
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
|
pip install -r requirements.txt
|
||||||
|
|
||||||
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
|
if [ "${WORKER_CUDA_VERSION}" = "11.8.0" ]; then \
|
||||||
|
pip install -U --force-reinstall torch==2.1.2 xformers==0.0.23.post1 --index-url https://download.pytorch.org/whl/cu118; \
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Copy built files from the build stage
|
||||||
|
COPY --from=build /vllm-installation/vllm/*.so /vllm-installation/vllm/
|
||||||
|
COPY vllm-${WORKER_CUDA_VERSION}/vllm vllm
|
||||||
|
|
||||||
|
# Set PYTHONPATH environment variable
|
||||||
|
ENV PYTHONPATH="/"
|
||||||
|
|
||||||
|
# Validate the installation
|
||||||
|
RUN python3 -c "import sys; print(sys.path); import vllm; print(vllm.__file__)"
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
This directory is for building the vllm-base image utilized by the worker.
|
||||||
@@ -0,0 +1,12 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
|
||||||
|
git clone https://github.com/runpod/vllm-fork-for-sls-worker.git
|
||||||
|
|
||||||
|
cp -r vllm-fork-for-sls-worker vllm-12.1.0
|
||||||
|
cp -r vllm-fork-for-sls-worker vllm-11.8.0
|
||||||
|
rm -rf vllm-fork-for-sls-worker
|
||||||
|
|
||||||
|
cd vllm-11.8.0
|
||||||
|
git checkout cuda11.8
|
||||||
|
|
||||||
|
echo "vLLM Base Image Builder Setup Complete."
|
||||||
Reference in New Issue
Block a user