Compare commits

...
23 Commits
Author SHA1 Message Date
alpayariyak 0e1e38326a Fix deprecated max_context_len_to_capture engine argument 2024-06-13 17:48:05 +00:00
alpayariyak c8458fef2b preparation for 1.0.0 release 2024-06-12 12:28:03 -07:00
alpayariyak bad5ddd892 Fix vLLM 0.4.1 bug for OpenAI /models route 2024-06-10 20:38:49 +00:00
alpayariyak f19ce12ab0 Fix Building Docker with model built-in #71 2024-06-07 15:29:46 -07:00
alpayariyak 1bb6f84541 Deprecated kv cache dtype warning 2024-05-10 16:29:20 +00:00
30cb56a3df Fix MODEL_REVISION env var (Merge pull request #50 from joennlae/rename-revision and #67 from mikljohansson/main)
Fixed MODEL_REVISION environment variable

Co-Authored-By: Jannis Schönleber <1493766+joennlae@users.noreply.github.com>
Co-Authored-By: Mikael Johansson <mikl.johansson@gmail.com>
2024-05-09 20:45:07 -04:00
alpayariyak 00add8707a Temporarily disable build and push github actions 2024-05-09 20:33:52 -04:00
alpayariyak ec7ea0b760 Update default base image version to fix github actions build 2024-05-09 20:18:34 -04:00
alpayariyak 9f2cb7b1d0 Update documentation to include rename of fp8_e5m2 to fp8 2024-05-09 23:59:11 +00:00
alpayariyak 4abe494635 Fix hf-transfer error 2024-05-09 23:51:07 +00:00
Alpay AriyakandGitHub 4f61b04afe Update README.md 2024-05-09 01:05:04 -04:00
Alpay Ariyakandalpayariyak 874379a0c5 1.0.0preview update for Llama 3 support and more (vLLM 0.3.3 -> 0.4.2) (#62) 2024-05-09 00:42:08 -04:00
Mikael Johansson f06a64d5b9 Fixed MODEL_REVISION environment variable 2024-05-02 13:45:43 +02:00
alpayariyak 0a5b5bc095 Add badges 2024-03-15 21:23:52 -04:00
alpayariyak 2936e4d95d Update automatic builds, documentation 2024-03-15 11:57:38 -04:00
Alpay AriyakandGitHub cee4e484d5 Update README.md for 0.3.2 2024-03-12 19:07:37 -04:00
Alpay AriyakandGitHub 6160769996 Release 0.3.2 2024-03-12 17:44:47 -05:00
alpayariyak d25b6f9628 Fix sampling params 2024-03-12 22:15:57 +00:00
alpayariyak c8ee100d80 Small refactor 2024-03-06 17:08:57 +00:00
alpayariyak fee8d8eee4 Fix submodule 2024-03-05 19:17:53 +00:00
alpayariyak db7167d57f 0.3.3 2024-03-05 19:14:35 +00:00
Alpay Ariyakandalpayariyak d91ccb866f 0.3.1: bug fixes 2024-02-29 02:55:44 -05:00
Alpay AriyakandGitHub 36e9b670ee Add notice on what to do when HuggingFace is down 2024-02-28 17:10:39 -05:00
19 changed files with 367 additions and 330 deletions
@@ -1,45 +0,0 @@
name: CD | Docker-Build-Release
on:
push:
branches:
- "main"
release:
types: [published]
workflow_dispatch:
inputs:
image_tag:
description: "Docker Image Tag"
required: false
default: "dev"
jobs:
docker-build:
runs-on: DO
# DO is a custom runner deployed on DigitalOcean, only available for workflows under the runpod-workers organization.
# If you would like to use this workflow, you can replace DO with ubuntu-latest or any other runner.
strategy:
matrix:
cuda_version: [11.8.0, 12.1.0]
steps:
- name: Set up QEMU
uses: docker/setup-qemu-action@v2
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v2
- name: Login to Docker Hub
uses: docker/login-action@v2
with:
username: ${{ secrets.DOCKERHUB_USERNAME }}
password: ${{ secrets.DOCKERHUB_TOKEN }}
# Build and push step
- name: Build and push
uses: docker/build-push-action@v4
with:
push: true
tags: ${{ vars.DOCKERHUB_REPO }}/${{ vars.DOCKERHUB_IMG }}:${{ (github.event_name == 'release' && github.event.release.tag_name) || (github.event_name == 'workflow_dispatch' && github.event.inputs.image_tag) || 'dev' }}-cuda${{ matrix.cuda_version }}
build-args: WORKER_CUDA_VERSION=${{ matrix.cuda_version }}
+3
View File
@@ -0,0 +1,3 @@
[submodule "vllm-base-image/vllm"]
path = vllm-base-image/vllm
url = https://github.com/runpod/vllm-fork-for-sls-worker.git
+8 -6
View File
@@ -1,5 +1,6 @@
ARG WORKER_CUDA_VERSION=11.8.0
FROM runpod/worker-vllm:base-0.3.0-cuda${WORKER_CUDA_VERSION} AS vllm-base
ARG BASE_IMAGE_VERSION=1.0.0
FROM runpod/worker-vllm:base-${BASE_IMAGE_VERSION}-cuda${WORKER_CUDA_VERSION} AS vllm-base
RUN apt-get update -y \
&& apt-get install -y python3-pip
@@ -19,7 +20,7 @@ ARG MODEL_REVISION=""
ARG TOKENIZER_REVISION=""
ENV MODEL_NAME=$MODEL_NAME \
MODEL_REVISION=$REVISION \
MODEL_REVISION=$MODEL_REVISION \
TOKENIZER_NAME=$TOKENIZER_NAME \
TOKENIZER_REVISION=$TOKENIZER_REVISION \
BASE_PATH=$BASE_PATH \
@@ -27,11 +28,11 @@ ENV MODEL_NAME=$MODEL_NAME \
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
HUGGINGFACE_HUB_CACHE="${BASE_PATH}/huggingface-cache/hub" \
HF_HOME="${BASE_PATH}/huggingface-cache/hub" \
HF_TRANSFER=1
HF_HUB_ENABLE_HF_TRANSFER=1
ENV PYTHONPATH="/:/vllm-installation"
ENV PYTHONPATH="/:/vllm-workspace"
COPY builder/download_model.py /download_model.py
COPY src/download_model.py /download_model.py
RUN --mount=type=secret,id=HF_TOKEN,required=false \
if [ -f /run/secrets/HF_TOKEN ]; then \
export HF_TOKEN=$(cat /run/secrets/HF_TOKEN); \
@@ -42,7 +43,8 @@ RUN --mount=type=secret,id=HF_TOKEN,required=false \
# Add source files
COPY src /src
# Remove download_model.py
RUN rm /download_model.py
# Start the handler
CMD ["python3", "/src/handler.py"]
+49 -28
View File
@@ -1,25 +1,34 @@
<div align="center">
<h1> vLLM Serverless Endpoint Worker </h1>
# OpenAI-Compatible vLLM Serverless Endpoint Worker
Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https://github.com/vllm-project/vllm) Inference Engine on RunPod Serverless with just a few clicks.
<!--
![vLLM Version](https://img.shields.io/badge/dynamic/yaml?url=https%3A%2F%2Fraw.githubusercontent.com%2Frunpod-workers%2Fworker-vllm%2Fmain%2Fvllm-base-image%2Fvllm-metadata.yml&query=%24.version&style=for-the-badge&logo=data%3Aimage%2Fsvg%2Bxml%3Bbase64%2CPD94bWwgdmVyc2lvbj0iMS4wIiBlbmNvZGluZz0iVVRGLTgiPz4KPCFET0NUWVBFIHN2ZyBQVUJMSUMgIi0vL1czQy8vRFREIFNWRyAxLjEvL0VOIiAiaHR0cDovL3d3dy53My5vcmcvR3JhcGhpY3MvU1ZHLzEuMS9EVEQvc3ZnMTEuZHRkIj4KPHN2ZyB4bWxucz0iaHR0cDovL3d3dy53My5vcmcvMjAwMC9zdmciIHZlcnNpb249IjEuMSIgd2lkdGg9IjU1cHgiIGhlaWdodD0iNTZweCIgc3R5bGU9InNoYXBlLXJlbmRlcmluZzpnZW9tZXRyaWNQcmVjaXNpb247IHRleHQtcmVuZGVyaW5nOmdlb21ldHJpY1ByZWNpc2lvbjsgaW1hZ2UtcmVuZGVyaW5nOm9wdGltaXplUXVhbGl0eTsgZmlsbC1ydWxlOmV2ZW5vZGQ7IGNsaXAtcnVsZTpldmVub2RkIiB4bWxuczp4bGluaz0iaHR0cDovL3d3dy53My5vcmcvMTk5OS94bGluayI%2BCjxnPjxwYXRoIHN0eWxlPSJvcGFjaXR5OjEiIGZpbGw9IiMzN2E0ZmUiIGQ9Ik0gNTEuNSwwLjUgQyA0Ni41ODIyLDE4LjA4MzggNDEuOTE1NiwzNS43NTA1IDM3LjUsNTMuNUMgMzIuMTY2Nyw1My41IDI2LjgzMzMsNTMuNSAyMS41LDUzLjVDIDIwLjgzMzMsNTMuNSAyMC41LDUzLjE2NjcgMjAuNSw1Mi41QyAyMS4zMzgyLDUyLjE1ODMgMjEuNjcxNiw1MS40OTE2IDIxLjUsNTAuNUMgMjIuMjIyOSw0Ni44NTU1IDIzLjIyMjksNDMuMTg4OSAyNC41LDM5LjVDIDI0LjY5MTcsMzYuMzk5MiAyNS4zNTg0LDMzLjM5OTIgMjYuNSwzMC41QyAyNi4yOTA3LDI5LjkxNCAyNS45NTc0LDI5LjQxNCAyNS41LDI5QyAyNy40NDE0LDI3LjE4NDEgMjguMTA4MSwyNS4xODQxIDI3LjUsMjNDIDI5LjI0MTUsMTguNTM4NyAzMC45MDgyLDE0LjAzODcgMzIuNSw5LjVDIDM4Ljc3NTcsNi4xOTM1OCA0NS4xMDkxLDMuMTkzNTggNTEuNSwwLjUgWiIvPjwvZz4KPGc%2BPHBhdGggc3R5bGU9Im9wYWNpdHk6MC45ODQiIGZpbGw9IiNmY2I3MWQiIGQ9Ik0gMjIuNSwxMi41IEMgMjEuNTA0NiwyNC45ODkgMjEuMTcxMywzNy42NTU3IDIxLjUsNTAuNUMgMjEuNjcxNiw1MS40OTE2IDIxLjMzODIsNTIuMTU4MyAyMC41LDUyLjVDIDEzLjAzMTEsMzkuMjI4NyA2LjM2NDQxLDI1LjU2MjEgMC41LDExLjVDIDguMDE5MDUsMTEuMTc1IDE1LjM1MjQsMTEuNTA4NCAyMi41LDEyLjUgWiIvPjwvZz4KPGc%2BPHBhdGggc3R5bGU9Im9wYWNpdHk6MC4wMiIgZmlsbD0iI2Q3ZGZlOCIgZD0iTSAyMi41LDEyLjUgQyAyMy4xNjY3LDIxLjUgMjMuODMzMywzMC41IDI0LjUsMzkuNUMgMjMuMjIyOSw0My4xODg5IDIyLjIyMjksNDYuODU1NSAyMS41LDUwLjVDIDIxLjE3MTMsMzcuNjU1NyAyMS41MDQ2LDI0Ljk4OSAyMi41LDEyLjUgWiIvPjwvZz4KPGc%2BPHBhdGggc3R5bGU9Im9wYWNpdHk6MC43NTMiIGZpbGw9IiNjZmQ2ZGQiIGQ9Ik0gNTEuNSwwLjUgQyA1Mi42MTI5LDEuOTQ2MzkgNTIuNzc5NiwzLjYxMzA1IDUyLDUuNUMgNDcuODAzNiwyMi4yODg3IDQzLjMwMzYsMzguOTU1MyAzOC41LDU1LjVDIDMyLjUsNTUuNSAyNi41LDU1LjUgMjAuNSw1NS41QyAyMC44MzMzLDU0LjgzMzMgMjEuMTY2Nyw1NC4xNjY3IDIxLjUsNTMuNUMgMjYuODMzMyw1My41IDMyLjE2NjcsNTMuNSAzNy41LDUzLjVDIDQxLjkxNTYsMzUuNzUwNSA0Ni41ODIyLDE4LjA4MzggNTEuNSwwLjUgWiIvPjwvZz4KPC9zdmc%2BCg%3D%3D&label=STABLE%20vLLM%20Version&link=https%3A%2F%2Fgithub.com%2Fvllm-project%2Fvllm)
![Worker Version](https://img.shields.io/github/v/tag/runpod-workers/worker-vllm?style=for-the-badge&logo=data%3Aimage%2Fsvg%2Bxml%3Bbase64%2CPD94bWwgdmVyc2lvbj0iMS4wIiBlbmNvZGluZz0idXRmLTgiPz4KPCEtLSBHZW5lcmF0b3I6IEFkb2JlIElsbHVzdHJhdG9yIDI2LjUuMywgU1ZHIEV4cG9ydCBQbHVnLUluIC4gU1ZHIFZlcnNpb246IDYuMDAgQnVpbGQgMCkgIC0tPgo8c3ZnIHZlcnNpb249IjEuMSIgaWQ9IkxheWVyXzEiIHhtbG5zPSJodHRwOi8vd3d3LnczLm9yZy8yMDAwL3N2ZyIgeG1sbnM6eGxpbms9Imh0dHA6Ly93d3cudzMub3JnLzE5OTkveGxpbmsiIHg9IjBweCIgeT0iMHB4IgoJIHZpZXdCb3g9IjAgMCAyMDAwIDIwMDAiIHN0eWxlPSJlbmFibGUtYmFja2dyb3VuZDpuZXcgMCAwIDIwMDAgMjAwMDsiIHhtbDpzcGFjZT0icHJlc2VydmUiPgo8c3R5bGUgdHlwZT0idGV4dC9jc3MiPgoJLnN0MHtmaWxsOiM2NzNBQjc7fQo8L3N0eWxlPgo8Zz4KCTxnPgoJCTxwYXRoIGNsYXNzPSJzdDAiIGQ9Ik0xMDE3Ljk1LDcxMS4wNGMtNC4yMiwyLjM2LTkuMTgsMy4wMS0xMy44NiwxLjgyTDM4Ni4xNyw1NTUuM2MtNDEuNzItMTAuNzYtODYuMDItMC42My0xMTYuNiwyOS43MwoJCQlsLTEuNCwxLjM5Yy0zNS45MiwzNS42NS0yNy41NSw5NS44LDE2Ljc0LDEyMC4zbDU4NC4zMiwzMjQuMjNjMzEuMzYsMTcuNCw1MC44Miw1MC40NSw1MC44Miw4Ni4zMnY4MDYuNzYKCQkJYzAsMzUuNDktMzguNDEsNTcuNjctNjkuMTUsMzkuOTRsLTcwMy4xNS00MDUuNjRjLTIzLjYtMTMuNjEtMzguMTMtMzguNzgtMzguMTMtNjYuMDJWNjY2LjYzYzAtODcuMjQsNDYuNDUtMTY3Ljg5LDEyMS45Mi0yMTEuNjYKCQkJTDkzMy44NSw0Mi4xNWMyMy40OC0xMy44LDUxLjQ3LTE3LjcsNzcuODMtMTAuODRsNzQ1LjcxLDE5NC4xYzMxLjUzLDguMjEsMzYuOTksNTAuNjUsOC41Niw2Ni41N0wxMDE3Ljk1LDcxMS4wNHoiLz4KCQk8cGF0aCBjbGFzcz0ic3QwIiBkPSJNMTUyNy43NSw1MzYuMzhsMTI4Ljg5LTc5LjYzbDE4OS45MiwxMDkuMTdjMjcuMjQsMTUuNjYsNDMuOTcsNDQuNzMsNDMuODIsNzYuMTVsLTQsODU3LjYKCQkJYy0wLjExLDI0LjM5LTEzLjE1LDQ2Ljg5LTM0LjI1LDU5LjExbC03MDEuNzUsNDA2LjYxYy0zMi4zLDE4LjcxLTcyLjc0LTQuNTktNzIuNzQtNDEuOTJ2LTc5Ny40MwoJCQljMC0zOC45OCwyMS4wNi03NC45MSw1NS4wNy05My45Nmw1OTAuMTctMzMwLjUzYzE4LjIzLTEwLjIxLDE4LjY1LTM2LjMsMC43NS00Ny4wOUwxNTI3Ljc1LDUzNi4zOHoiLz4KCQk8cGF0aCBjbGFzcz0ic3QwIiBkPSJNMTUyNC4wMSw2NjUuOTEiLz4KCTwvZz4KPC9nPgo8L3N2Zz4K&logoColor=%23ffffff&label=STABLE%20Worker%20Version&color=%23673ab7)
![vLLM Version](https://img.shields.io/badge/dynamic/yaml?url=https%3A%2F%2Fraw.githubusercontent.com%2Frunpod-workers%2Fworker-vllm%2Fmain%2Fvllm-base-image%2Fvllm-metadata.yml&query=%24.dev_version&style=for-the-badge&logo=data%3Aimage%2Fsvg%2Bxml%3Bbase64%2CPD94bWwgdmVyc2lvbj0iMS4wIiBlbmNvZGluZz0iVVRGLTgiPz4KPCFET0NUWVBFIHN2ZyBQVUJMSUMgIi0vL1czQy8vRFREIFNWRyAxLjEvL0VOIiAiaHR0cDovL3d3dy53My5vcmcvR3JhcGhpY3MvU1ZHLzEuMS9EVEQvc3ZnMTEuZHRkIj4KPHN2ZyB4bWxucz0iaHR0cDovL3d3dy53My5vcmcvMjAwMC9zdmciIHZlcnNpb249IjEuMSIgd2lkdGg9IjU1cHgiIGhlaWdodD0iNTZweCIgc3R5bGU9InNoYXBlLXJlbmRlcmluZzpnZW9tZXRyaWNQcmVjaXNpb247IHRleHQtcmVuZGVyaW5nOmdlb21ldHJpY1ByZWNpc2lvbjsgaW1hZ2UtcmVuZGVyaW5nOm9wdGltaXplUXVhbGl0eTsgZmlsbC1ydWxlOmV2ZW5vZGQ7IGNsaXAtcnVsZTpldmVub2RkIiB4bWxuczp4bGluaz0iaHR0cDovL3d3dy53My5vcmcvMTk5OS94bGluayI%2BCjxnPjxwYXRoIHN0eWxlPSJvcGFjaXR5OjEiIGZpbGw9IiMzN2E0ZmUiIGQ9Ik0gNTEuNSwwLjUgQyA0Ni41ODIyLDE4LjA4MzggNDEuOTE1NiwzNS43NTA1IDM3LjUsNTMuNUMgMzIuMTY2Nyw1My41IDI2LjgzMzMsNTMuNSAyMS41LDUzLjVDIDIwLjgzMzMsNTMuNSAyMC41LDUzLjE2NjcgMjAuNSw1Mi41QyAyMS4zMzgyLDUyLjE1ODMgMjEuNjcxNiw1MS40OTE2IDIxLjUsNTAuNUMgMjIuMjIyOSw0Ni44NTU1IDIzLjIyMjksNDMuMTg4OSAyNC41LDM5LjVDIDI0LjY5MTcsMzYuMzk5MiAyNS4zNTg0LDMzLjM5OTIgMjYuNSwzMC41QyAyNi4yOTA3LDI5LjkxNCAyNS45NTc0LDI5LjQxNCAyNS41LDI5QyAyNy40NDE0LDI3LjE4NDEgMjguMTA4MSwyNS4xODQxIDI3LjUsMjNDIDI5LjI0MTUsMTguNTM4NyAzMC45MDgyLDE0LjAzODcgMzIuNSw5LjVDIDM4Ljc3NTcsNi4xOTM1OCA0NS4xMDkxLDMuMTkzNTggNTEuNSwwLjUgWiIvPjwvZz4KPGc%2BPHBhdGggc3R5bGU9Im9wYWNpdHk6MC45ODQiIGZpbGw9IiNmY2I3MWQiIGQ9Ik0gMjIuNSwxMi41IEMgMjEuNTA0NiwyNC45ODkgMjEuMTcxMywzNy42NTU3IDIxLjUsNTAuNUMgMjEuNjcxNiw1MS40OTE2IDIxLjMzODIsNTIuMTU4MyAyMC41LDUyLjVDIDEzLjAzMTEsMzkuMjI4NyA2LjM2NDQxLDI1LjU2MjEgMC41LDExLjVDIDguMDE5MDUsMTEuMTc1IDE1LjM1MjQsMTEuNTA4NCAyMi41LDEyLjUgWiIvPjwvZz4KPGc%2BPHBhdGggc3R5bGU9Im9wYWNpdHk6MC4wMiIgZmlsbD0iI2Q3ZGZlOCIgZD0iTSAyMi41LDEyLjUgQyAyMy4xNjY3LDIxLjUgMjMuODMzMywzMC41IDI0LjUsMzkuNUMgMjMuMjIyOSw0My4xODg5IDIyLjIyMjksNDYuODU1NSAyMS41LDUwLjVDIDIxLjE3MTMsMzcuNjU1NyAyMS41MDQ2LDI0Ljk4OSAyMi41LDEyLjUgWiIvPjwvZz4KPGc%2BPHBhdGggc3R5bGU9Im9wYWNpdHk6MC43NTMiIGZpbGw9IiNjZmQ2ZGQiIGQ9Ik0gNTEuNSwwLjUgQyA1Mi42MTI5LDEuOTQ2MzkgNTIuNzc5NiwzLjYxMzA1IDUyLDUuNUMgNDcuODAzNiwyMi4yODg3IDQzLjMwMzYsMzguOTU1MyAzOC41LDU1LjVDIDMyLjUsNTUuNSAyNi41LDU1LjUgMjAuNSw1NS41QyAyMC44MzMzLDU0LjgzMzMgMjEuMTY2Nyw1NC4xNjY3IDIxLjUsNTMuNUMgMjYuODMzMyw1My41IDMyLjE2NjcsNTMuNSAzNy41LDUzLjVDIDQxLjkxNTYsMzUuNzUwNSA0Ni41ODIyLDE4LjA4MzggNTEuNSwwLjUgWiIvPjwvZz4KPC9zdmc%2BCg%3D%3D&label=DEV%20vLLM%20Version%20&link=https%3A%2F%2Fgithub.com%2Fvllm-project%2Fvllm)\
![Docker Pulls](https://img.shields.io/docker/pulls/runpod/worker-vllm?style=for-the-badge&logo=docker&label=Docker%20Pulls&link=https%3A%2F%2Fhub.docker.com%2Frepository%2Fdocker%2Frunpod%2Fworker-vllm%2Fgeneral) -->
<!--
![Docker Automatic Build](https://img.shields.io/github/actions/workflow/status/runpod-workers/worker-vllm/docker-build-release.yml?style=flat&label=BUILD) -->
[![CD | Docker-Build-Release](https://github.com/runpod-workers/worker-vllm/actions/workflows/docker-build-release.yml/badge.svg)](https://github.com/runpod-workers/worker-vllm/actions/workflows/docker-build-release.yml)
Deploy Blazing-fast LLMs powered by [vLLM](https://github.com/vllm-project/vllm) on RunPod Serverless in a few clicks.
</div>
### Worker vLLM 0.3.0: What's New since 0.2.0:
- **🚀 Full OpenAI Compatibility 🚀**
# News:
### 1. UI for Deploying vLLM Worker on RunPod console:
![Demo of Deploying vLLM Worker on RunPod console with new UI](media/ui_demo.gif)
### 2. Worker vLLM `1.0.0` with vLLM `0.4.2` now available under `stable` tags
Update 1.0.0 is now available, use the image tag `runpod/worker-vllm:stable-cuda12.1.0` or `runpod/worker-vllm:stable-cuda11.8.0`.
### 3. OpenAI-Compatible [Embedding Worker](https://github.com/runpod-workers/worker-infinity-embedding) Released
Deploy your own OpenAI-compatible Serverless Endpoint on RunPod with multiple embedding models and fast inference for RAG and more!
### 4. Caching Accross RunPod Machines
Worker vLLM is now cached on all RunPod machines, resulting in near-instant deployment! Previously, downloading and extracting the image took 3-5 minutes on average.
You may now use your deployment with any OpenAI Codebase by changing **only 3 lines** in total. The supported routes are <ins>Chat Completions</ins>, <ins>Completions</ins>, and <ins>Models</ins> - with both streaming and non-streaming.
- **Dynamic Batch Size** - time-to-first token as fast no batching, while maintaining the performance of batched token streaming throughout the request.
- vLLM 0.2.7 -> 0.3.2
- Gemma, DeepSeek MoE and OLMo support.
- FP8 KV Cache support
- New supported parameters
- We're working on adding support for Multi-LoRA ⚙️
- Support for a wide range of new settings for your endpoint, such as Custom chat templates.
- Fixed Tensor Parallelism, baking model into images, and more bugs.
- Refactors and general improvements.
## Table of Contents
- [Setting up the Serverless Worker](#setting-up-the-serverless-worker)
@@ -53,8 +62,11 @@ Deploy Blazing-fast LLMs powered by [vLLM](https://github.com/vllm-project/vllm)
# Setting up the Serverless Worker
### Option 1: Deploy Any Model Using Pre-Built Docker Image [Recommended]
> [!TIP]
> This is the recommended way to deploy your model, as it does not require you to build a Docker image, upload heavy models to DockerHub and wait for workers to download them. Instead, use this option to deploy your model in a few clicks. For even more convenience, attach a network storage volume to your Endpoint, which will download the model once and share it across all workers.
> [!NOTE]
> You can now deploy from the dedicated UI on the RunPod console with all of the settings and choices listed.
> Try now by accessing in Explore or Serverless pages on the RunPod console!
We now offer a pre-built Docker Image for the vLLM Worker that you can configure entirely with Environment Variables when creating the RunPod Serverless Endpoint:
@@ -66,17 +78,17 @@ Below is a summary of the available RunPod Worker images, categorized by image s
| CUDA Version | Stable Image Tag | Development Image Tag | Note |
|--------------|-----------------------------------|-----------------------------------|----------------------------------------------------------------------|
| 11.8.0 | `runpod/worker-vllm:0.3.0-cuda11.8.0` | `runpod/worker-vllm:dev-cuda11.8.0` | Available on all RunPod Workers without additional selection needed. |
| 12.1.0 | `runpod/worker-vllm:0.3.0-cuda12.1.0` | `runpod/worker-vllm:dev-cuda12.1.0` | When creating an Endpoint, select CUDA Version 12.2 and 12.1 in the filter. |
| 11.8.0 | `runpod/worker-vllm:stable-cuda11.8.0` | `runpod/worker-vllm:dev-cuda11.8.0` | Available on all RunPod Workers without additional selection needed. |
| 12.1.0 | `runpod/worker-vllm:stable-cuda12.1.0` | `runpod/worker-vllm:dev-cuda12.1.0` | When creating an Endpoint, select CUDA Version 12.3, 12.2 and 12.1 in the filter. |
This table provides a quick reference to the image tags you should use based on the desired CUDA version and image stability (Stable or Development). Ensure to follow the selection note for CUDA 12.1.0 compatibility.
---
#### Prerequisites
- RunPod Account
#### Environment Variables
#### Environment Variables/Settings
> Note: `0` is equivalent to `False` and `1` is equivalent to `True` for boolean values.
| Name | Default | Type/Choices | Description |
@@ -84,14 +96,14 @@ This table provides a quick reference to the image tags you should use based on
**LLM Settings**
| `MODEL_NAME`**\*** | - | `str` | Hugging Face Model Repository (e.g., `openchat/openchat-3.5-1210`). |
| `MODEL_REVISION` | `None` | `str` |Model revision(branch) to load. |
| `MAX_MODEL_LENGTH` | Model's maximum | `int` |Maximum number of tokens for the engine to handle per request. |
| `MAX_MODEL_LEN` | Model's maximum | `int` |Maximum number of tokens for the engine to handle per request. |
| `BASE_PATH` | `/runpod-volume` | `str` |Storage directory for Huggingface cache and model. Utilizes network storage if attached when pointed at `/runpod-volume`, which will have only one worker download the model once, which all workers will be able to load. If no network volume is present, creates a local directory within each worker. |
| `LOAD_FORMAT` | `auto` | `str` |Format to load model in. |
| `HF_TOKEN` | - | `str` |Hugging Face token for private and gated models. |
| `QUANTIZATION` | `None` | `awq`, `squeezellm`, `gptq` |Quantization of given model. The model must already be quantized. |
| `TRUST_REMOTE_CODE` | `0` | boolean as `int` |Trust remote code for Hugging Face models. Can help with Mixtral 8x7B, Quantized models, and unusual models/architectures.
| `SEED` | `0` | `int` |Sets random seed for operations. |
| `KV_CACHE_DTYPE` | `auto` | boolean as `int` |Data type for kv cache storage. Uses `DTYPE` if set to `auto`. |
| `KV_CACHE_DTYPE` | `auto` | `auto`, `fp8` |Data type for kv cache storage. Uses `DTYPE` if set to `auto`. |
| `DTYPE` | `auto` | `auto`, `half`, `float16`, `bfloat16`, `float`, `float32` |Sets datatype/precision for model weights and activations. |
**Tokenizer Settings**
| `TOKENIZER_NAME` | `None` | `str` |Tokenizer repository to use a different tokenizer than the model's default. |
@@ -103,7 +115,7 @@ This table provides a quick reference to the image tags you should use based on
| `BLOCK_SIZE` | `16` | `8`, `16`, `32` |Token block size for contiguous chunks of tokens. |
| `SWAP_SPACE` | `4` | `int` |CPU swap space size (GiB) per GPU. |
| `ENFORCE_EAGER` | `0` | boolean as `int` |Always use eager-mode PyTorch. If False(`0`), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility. |
| `MAX_CONTEXT_LEN_TO_CAPTURE` | `8192` | `int` |Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode.|
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` |Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode.|
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` |Enables or disables custom all reduce. |
**Streaming Batch Size Settings**:
| `DEFAULT_BATCH_SIZE` | `50` | `int` |Default and Maximum batch size for token streaming to reduce HTTP calls. |
@@ -170,6 +182,8 @@ Below are all supported model architectures (and examples of each) that you can
- Baichuan & Baichuan2 (`baichuan-inc/Baichuan2-13B-Chat`, `baichuan-inc/Baichuan-7B`, etc.)
- BLOOM (`bigscience/bloom`, `bigscience/bloomz`, etc.)
- ChatGLM (`THUDM/chatglm2-6b`, `THUDM/chatglm3-6b`, etc.)
- Command-R (`CohereForAI/c4ai-command-r-v01`, etc.)
- DBRX (`databricks/dbrx-base`, `databricks/dbrx-instruct` etc.)
- DeciLM (`Deci/DeciLM-7B`, `Deci/DeciLM-7B-instruct`, etc.)
- Falcon (`tiiuae/falcon-7b`, `tiiuae/falcon-40b`, `tiiuae/falcon-rw-7b`, etc.)
- Gemma (`google/gemma-2b`, `google/gemma-7b`, etc.)
@@ -179,16 +193,23 @@ Below are all supported model architectures (and examples of each) that you can
- GPT-NeoX (`EleutherAI/gpt-neox-20b`, `databricks/dolly-v2-12b`, `stabilityai/stablelm-tuned-alpha-7b`, etc.)
- InternLM (`internlm/internlm-7b`, `internlm/internlm-chat-7b`, etc.)
- InternLM2 (`internlm/internlm2-7b`, `internlm/internlm2-chat-7b`, etc.)
- LLaMA & LLaMA-2 (`meta-llama/Llama-2-70b-hf`, `lmsys/vicuna-13b-v1.3`, `young-geng/koala`, `openlm-research/open_llama_13b`, etc.)
- Jais (`core42/jais-13b`, `core42/jais-13b-chat`, `core42/jais-30b-v3`, `core42/jais-30b-chat-v3`, etc.)
- LLaMA, Llama 2, and Meta Llama 3 (`meta-llama/Meta-Llama-3-8B-Instruct`, `meta-llama/Meta-Llama-3-70B-Instruct`, `meta-llama/Llama-2-70b-hf`, `lmsys/vicuna-13b-v1.3`, `young-geng/koala`, `openlm-research/open_llama_13b`, etc.)
- MiniCPM (`openbmb/MiniCPM-2B-sft-bf16`, `openbmb/MiniCPM-2B-dpo-bf16`, etc.)
- Mistral (`mistralai/Mistral-7B-v0.1`, `mistralai/Mistral-7B-Instruct-v0.1`, etc.)
- Mixtral (`mistralai/Mixtral-8x7B-v0.1`, `mistralai/Mixtral-8x7B-Instruct-v0.1`, etc.)
- Mixtral (`mistralai/Mixtral-8x7B-v0.1`, `mistralai/Mixtral-8x7B-Instruct-v0.1`, `mistral-community/Mixtral-8x22B-v0.1`, etc.)
- MPT (`mosaicml/mpt-7b`, `mosaicml/mpt-30b`, etc.)
- OLMo (`allenai/OLMo-1B`, `allenai/OLMo-7B`, etc.)
- OLMo (`allenai/OLMo-1B-hf`, `allenai/OLMo-7B-hf`, etc.)
- OPT (`facebook/opt-66b`, `facebook/opt-iml-max-30b`, etc.)
- Orion (`OrionStarAI/Orion-14B-Base`, `OrionStarAI/Orion-14B-Chat`, etc.)
- Phi (`microsoft/phi-1_5`, `microsoft/phi-2`, etc.)
- Phi-3 (`microsoft/Phi-3-mini-4k-instruct`, `microsoft/Phi-3-mini-128k-instruct`, etc.)
- Qwen (`Qwen/Qwen-7B`, `Qwen/Qwen-7B-Chat`, etc.)
- Qwen2 (`Qwen/Qwen2-7B-beta`, `Qwen/Qwen-7B-Chat-beta`, etc.)
- Qwen2 (`Qwen/Qwen1.5-7B`, `Qwen/Qwen1.5-7B-Chat`, etc.)
- Qwen2MoE (`Qwen/Qwen1.5-MoE-A2.7B`, `Qwen/Qwen1.5-MoE-A2.7B-Chat`, etc.)
- StableLM(`stabilityai/stablelm-3b-4e1t`, `stabilityai/stablelm-base-alpha-7b-v2`, etc.)
- Starcoder2(`bigcode/starcoder2-3b`, `bigcode/starcoder2-7b`, `bigcode/starcoder2-15b`, etc.)
- Xverse (`xverse/XVERSE-7B-Chat`, `xverse/XVERSE-13B-Chat`, `xverse/XVERSE-65B-Chat`, etc.)
- Yi (`01-ai/Yi-6B`, `01-ai/Yi-34B`, etc.)
# Usage: OpenAI Compatibility
-51
View File
@@ -1,51 +0,0 @@
import os
import shutil
from huggingface_hub import snapshot_download
from vllm.model_executor.weight_utils import prepare_hf_model_weights, Disabledtqdm
def download_extras_or_tokenizer(model_name, cache_dir, revision, extras=False):
"""Download model or tokenizer and prepare its weights, returning the local folder path."""
pattern = ["*token*", "*.json"] if extras else None
extra_dir = "/extras" if extras else ""
folder = snapshot_download(
model_name,
cache_dir=cache_dir + extra_dir,
revision=revision,
tqdm_class=Disabledtqdm,
allow_patterns=pattern if extras else None,
ignore_patterns=["*.safetensors", "*.bin", "*.pt"] if not extras else None
)
return folder
def move_files(src_dir, dest_dir):
"""Move files from source to destination directory."""
for f in os.listdir(src_dir):
src_path = os.path.join(src_dir, f)
dst_path = os.path.join(dest_dir, f)
shutil.copy2(src_path, dst_path)
os.remove(src_path)
if __name__ == "__main__":
model, download_dir = os.getenv("MODEL_NAME"), os.getenv("HF_HOME")
tokenizer = os.getenv("TOKENIZER_NAME") or model
revisions = {
"model": os.getenv("MODEL_REVISION") or None,
"tokenizer": os.getenv("TOKENIZER_REVISION") or None
}
if not model or not download_dir:
raise ValueError(f"Must specify model and download_dir. Model: {model}, download_dir: {download_dir}")
os.makedirs(download_dir, exist_ok=True)
model_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights(model_name_or_path=model, revision=revisions["model"], cache_dir=download_dir)
model_extras_folder = download_extras_or_tokenizer(model, download_dir, revisions["model"], extras=True)
move_files(model_extras_folder, model_folder)
with open("/local_model_path.txt", "w") as f:
f.write(model_folder)
if tokenizer != model:
tokenizer_folder = download_extras_or_tokenizer(tokenizer, download_dir, revisions["tokenizer"])
with open("/local_tokenizer_path.txt", "w") as f:
f.write(tokenizer_folder)
+3 -2
View File
@@ -1,4 +1,3 @@
hf_transfer
ray
pandas
pyarrow
@@ -6,4 +5,6 @@ runpod==1.6.2
huggingface-hub
packaging
typing-extensions==4.7.1
pydantic
pydantic
pydantic-settings
hf-transfer
+65
View File
@@ -0,0 +1,65 @@
variable "PUSH" {
default = "true"
}
variable "REPOSITORY" {
default = "runpod"
}
variable "BASE_IMAGE_VERSION" {
default = "1.0.0"
}
group "all" {
targets = ["base", "main"]
}
group "base" {
targets = ["base-1180", "base-1210"]
}
group "main" {
targets = ["worker-1180", "worker-1210"]
}
target "base-1180" {
tags = ["${REPOSITORY}/worker-vllm:base-${BASE_IMAGE_VERSION}-cuda11.8.0"]
context = "vllm-base-image"
dockerfile = "Dockerfile"
args = {
WORKER_CUDA_VERSION = "11.8.0"
}
output = ["type=docker,push=${PUSH}"]
}
target "base-1210" {
tags = ["${REPOSITORY}/worker-vllm:base-${BASE_IMAGE_VERSION}-cuda12.1.0"]
context = "vllm-base-image"
dockerfile = "Dockerfile"
args = {
WORKER_CUDA_VERSION = "12.1.0"
}
output = ["type=docker,push=${PUSH}"]
}
target "worker-1180" {
tags = ["${REPOSITORY}/worker-vllm:${BASE_IMAGE_VERSION}-cuda11.8.0"]
context = "."
dockerfile = "Dockerfile"
args = {
BASE_IMAGE_VERSION = "${BASE_IMAGE_VERSION}"
WORKER_CUDA_VERSION = "11.8.0"
}
output = ["type=docker,push=${PUSH}"]
}
target "worker-1210" {
tags = ["${REPOSITORY}/worker-vllm:${BASE_IMAGE_VERSION}-cuda12.1.0"]
context = "."
dockerfile = "Dockerfile"
args = {
BASE_IMAGE_VERSION = "${BASE_IMAGE_VERSION}"
WORKER_CUDA_VERSION = "12.1.0"
}
output = ["type=docker,push=${PUSH}"]
}
BIN
View File
Binary file not shown.

After

Width:  |  Height:  |  Size: 27 MiB

+36 -25
View File
@@ -1,27 +1,31 @@
import os
import json
import logging
from dotenv import load_dotenv
from utils import count_physical_cores
from torch.cuda import device_count
from utils import get_int_bool_env
class EngineConfig:
def __init__(self):
load_dotenv()
self.model_name_or_path, self.hf_home, self.model_revision = self._get_local_or_env("/local_model_path.txt", "MODEL_NAME")
self.tokenizer_name_or_path, _, self.tokenizer_revision = self._get_local_or_env("/local_tokenizer_path.txt", "TOKENIZER_NAME")
self.tokenizer_name_or_path = self.tokenizer_name_or_path or self.model_name_or_path
self.quantization = self._get_quantization()
self.hf_home = os.getenv("HF_HOME")
# Check if /local_metadata.json exists
local_metadata = {}
if os.path.exists("/local_metadata.json"):
with open("/local_metadata.json", "r") as f:
local_metadata = json.load(f)
if local_metadata.get("model_name") is None:
raise ValueError("Model name is not found in /local_metadata.json, there was a problem when you baked the model in.")
logging.info("Using baked-in model")
os.environ["TRANSFORMERS_OFFLINE"] = "1"
os.environ["HF_HUB_OFFLINE"] = "1"
self.model_name_or_path = local_metadata.get("model_name", os.getenv("MODEL_NAME"))
self.model_revision = local_metadata.get("revision", os.getenv("MODEL_REVISION"))
self.tokenizer_name_or_path = local_metadata.get("tokenizer_name", os.getenv("TOKENIZER_NAME")) or self.model_name_or_path
self.tokenizer_revision = local_metadata.get("tokenizer_revision", os.getenv("TOKENIZER_REVISION"))
self.quantization = local_metadata.get("quantization", os.getenv("QUANTIZATION"))
self.config = self._initialize_config()
def _get_local_or_env(self, local_path, env_var):
if os.path.exists(local_path):
with open(local_path, "r") as file:
return file.read().strip(), None, None
return os.getenv(env_var), os.getenv("HF_HOME"), os.getenv(f"{env_var}_REVISION")
def _get_quantization(self):
quantization = os.getenv("QUANTIZATION", "").lower()
return quantization if quantization in ["awq", "squeezellm", "gptq"] else None
def _initialize_config(self):
args = {
"model": self.model_name_or_path,
@@ -32,20 +36,27 @@ class EngineConfig:
"dtype": os.getenv("DTYPE", "half" if self.quantization else "auto"),
"tokenizer": self.tokenizer_name_or_path,
"tokenizer_revision": self.tokenizer_revision,
"disable_log_stats": bool(int(os.getenv("DISABLE_LOG_STATS", 1))),
"disable_log_requests": bool(int(os.getenv("DISABLE_LOG_REQUESTS", 1))),
"trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))),
"disable_log_stats": get_int_bool_env("DISABLE_LOG_STATS", True),
"disable_log_requests": get_int_bool_env("DISABLE_LOG_REQUESTS", True),
"trust_remote_code": get_int_bool_env("TRUST_REMOTE_CODE", False),
"gpu_memory_utilization": float(os.getenv("GPU_MEMORY_UTILIZATION", 0.95)),
"max_parallel_loading_workers": None if device_count() > 1 or not os.getenv("MAX_PARALLEL_LOADING_WORKERS") else int(os.getenv("MAX_PARALLEL_LOADING_WORKERS")),
"max_model_len": int(os.getenv("MAX_MODEL_LENGTH")) if os.getenv("MAX_MODEL_LENGTH") else None,
"max_model_len": int(os.getenv("MAX_MODEL_LEN")) if os.getenv("MAX_MODEL_LEN") else None,
"tensor_parallel_size": device_count(),
"seed": int(os.getenv("SEED")) if os.getenv("SEED") else None,
"kv_cache_dtype": os.getenv("KV_CACHE_DTYPE"),
"block_size": int(os.getenv("BLOCK_SIZE")) if os.getenv("BLOCK_SIZE") else None,
"swap_space": int(os.getenv("SWAP_SPACE")) if os.getenv("SWAP_SPACE") else None,
"max_context_len_to_capture": int(os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE")) if os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE") else None,
"disable_custom_all_reduce": bool(int(os.getenv("DISABLE_CUSTOM_ALL_REDUCE", 0))),
"enforce_eager": bool(int(os.getenv("ENFORCE_EAGER", 0)))
"max_seq_len_to_capture": int(os.getenv("MAX_SEQ_LEN_TO_CAPTURE")) if os.getenv("MAX_SEQ_LEN_TO_CAPTURE") else None,
"disable_custom_all_reduce": get_int_bool_env("DISABLE_CUSTOM_ALL_REDUCE", False),
"enforce_eager": get_int_bool_env("ENFORCE_EAGER", False)
}
return {k: v for k, v in args.items() if v is not None}
if args["kv_cache_dtype"] == "fp8_e5m2":
args["kv_cache_dtype"] = "fp8"
logging.warning("Using fp8_e5m2 is deprecated. Please use fp8 instead.")
if os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE"):
args["max_seq_len_to_capture"] = int(os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE"))
logging.warning("Using MAX_CONTEXT_LEN_TO_CAPTURE is deprecated. Please use MAX_SEQ_LEN_TO_CAPTURE instead.")
return {k: v for k, v in args.items() if v not in [None, ""]}
+1 -27
View File
@@ -1,30 +1,4 @@
from typing import Union
DEFAULT_BATCH_SIZE = 50
DEFAULT_MAX_CONCURRENCY = 300
DEFAULT_BATCH_SIZE_GROWTH_FACTOR = 3
DEFAULT_MIN_BATCH_SIZE = 1
SAMPLING_PARAM_TYPES = {
"n": int,
"best_of": int,
"presence_penalty": float,
"frequency_penalty": float,
"repetition_penalty": float,
"temperature": Union[float, int],
"top_p": float,
"top_k": int,
"min_p": float,
"use_beam_search": bool,
"length_penalty": float,
"early_stopping": Union[bool, str],
"stop": Union[str, list],
"stop_token_ids": list,
"ignore_eos": bool,
"max_tokens": int,
"logprobs": int,
"prompt_logprobs": int,
"skip_special_tokens": bool,
"spaces_between_special_tokens": bool,
"include_stop_str_in_output": bool
}
DEFAULT_MIN_BATCH_SIZE = 1
+27
View File
@@ -0,0 +1,27 @@
import os
from huggingface_hub import snapshot_download
import json
if __name__ == "__main__":
model_name = os.getenv("MODEL_NAME")
if not model_name:
raise ValueError("Must specify model name by adding --build-arg MODEL_NAME=<your model's repo>")
revision = os.getenv("MODEL_REVISION") or None
snapshot_download(model_name, revision=revision, cache_dir=os.getenv("HF_HOME"))
tokenizer_name = os.getenv("TOKENIZER_NAME") or None
tokenizer_revision = os.getenv("TOKENIZER_REVISION") or None
if tokenizer_name:
snapshot_download(tokenizer_name, revision=tokenizer_revision, cache_dir=os.getenv("HF_HOME"))
# Create file with metadata of baked in model and/or tokenizer
with open("/local_metadata.json", "w") as f:
json.dump({
"model_name": model_name,
"revision": revision,
"tokenizer_name": tokenizer_name or model_name,
"tokenizer_revision": tokenizer_revision or revision,
"quantization": os.getenv("QUANTIZATION")
}, f)
+12 -6
View File
@@ -5,8 +5,9 @@ import json
from dotenv import load_dotenv
from torch.cuda import device_count
from typing import AsyncGenerator
import time
from vllm import AsyncLLMEngine, AsyncEngineArgs, SamplingParams
from vllm import AsyncLLMEngine, AsyncEngineArgs
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
from vllm.entrypoints.openai.serving_completion import OpenAIServingCompletion
from vllm.entrypoints.openai.protocol import ChatCompletionRequest, CompletionRequest, ErrorResponse
@@ -16,7 +17,6 @@ from constants import DEFAULT_MAX_CONCURRENCY, DEFAULT_BATCH_SIZE, DEFAULT_BATCH
from tokenizer import TokenizerWrapper
from config import EngineConfig
class vLLMEngine:
def __init__(self, engine = None):
load_dotenv() # For local development
@@ -35,7 +35,7 @@ class vLLMEngine:
try:
async for batch in self._generate_vllm(
llm_input=job_input.llm_input,
validated_sampling_params=job_input.validated_sampling_params,
validated_sampling_params=job_input.sampling_params,
batch_size=job_input.max_batch_size,
stream=job_input.stream,
apply_chat_template=job_input.apply_chat_template,
@@ -45,12 +45,11 @@ class vLLMEngine:
):
yield batch
except Exception as e:
yield create_error_response(str(e)).model_dump()
yield {"error": create_error_response(str(e)).model_dump()}
async def _generate_vllm(self, llm_input, validated_sampling_params, batch_size, stream, apply_chat_template, request_id, batch_size_growth_factor, min_batch_size: str) -> AsyncGenerator[dict, None]:
if apply_chat_template or isinstance(llm_input, list):
llm_input = self.tokenizer.apply_chat_template(llm_input)
validated_sampling_params = SamplingParams(**validated_sampling_params)
results_generator = self.llm.generate(llm_input, validated_sampling_params, request_id)
n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True
last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0}
@@ -102,7 +101,11 @@ class vLLMEngine:
def _initialize_llm(self):
try:
return AsyncLLMEngine.from_engine_args(AsyncEngineArgs(**self.config))
start = time.time()
engine = AsyncLLMEngine.from_engine_args(AsyncEngineArgs(**self.config))
end = time.time()
logging.info(f"Initialized vLLM engine in {end - start:.2f}s")
return engine
except Exception as e:
logging.error("Error initializing vLLM engine: %s", e)
raise e
@@ -138,6 +141,9 @@ class OpenAIvLLMEngine:
async def _handle_model_request(self):
models = await self.chat_engine.show_available_models()
fixed_model = models.data[0]
fixed_model.id = self.served_model_name
models.data = [fixed_model]
return models.model_dump()
async def _handle_chat_or_completion_request(self, openai_request: JobInput):
+11 -19
View File
@@ -1,10 +1,9 @@
import os
import logging
from http import HTTPStatus
from typing import Any, Dict
from constants import SAMPLING_PARAM_TYPES
from vllm.utils import random_uuid
from vllm.entrypoints.openai.protocol import ErrorResponse
from vllm import SamplingParams
logging.basicConfig(level=logging.INFO)
@@ -25,20 +24,6 @@ def count_physical_cores():
return len(cores)
def validate_sampling_params(params: Dict[str, Any]) -> Dict[str, Any]:
validated_params = {}
invalid_params = []
for key, value in params.items():
expected_type = SAMPLING_PARAM_TYPES.get(key)
if expected_type and isinstance(value, expected_type):
validated_params[key] = value
else:
invalid_params.append(key)
if len(invalid_params) > 0:
logging.warning("Ignoring invalid sampling params: %s", invalid_params)
return validated_params
class JobInput:
def __init__(self, job):
@@ -47,7 +32,7 @@ class JobInput:
self.max_batch_size = job.get("max_batch_size")
self.apply_chat_template = job.get("apply_chat_template", False)
self.use_openai_format = job.get("use_openai_format", False)
self.validated_sampling_params = validate_sampling_params(job.get("sampling_params", {}))
self.sampling_params = SamplingParams(**job.get("sampling_params", {}))
self.request_id = random_uuid()
batch_size_growth_factor = job.get("batch_size_growth_factor")
self.batch_size_growth_factor = float(batch_size_growth_factor) if batch_size_growth_factor else None
@@ -78,4 +63,11 @@ class BatchSize:
def create_error_response(message: str, err_type: str = "BadRequestError", status_code: HTTPStatus = HTTPStatus.BAD_REQUEST) -> ErrorResponse:
return ErrorResponse(message=message,
type=err_type,
code=status_code.value)
code=status_code.value)
def get_int_bool_env(env_var: str, default: bool) -> bool:
return int(os.getenv(env_var, int(default))) == 1
+149
View File
@@ -0,0 +1,149 @@
################### vLLM Base Dockerfile ###################
# This Dockerfile is for building the image that the
# vLLM worker container will use as its base image.
# If your changes are outside of the vLLM source code, you
# do not need to build this image.
##########################################################
# Define the CUDA version for the build
ARG WORKER_CUDA_VERSION=11.8.0
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-devel-ubuntu22.04 AS dev
# Re-declare ARG after FROM
ARG WORKER_CUDA_VERSION
# Update and install dependencies
RUN apt-get update -y \
&& apt-get install -y python3-pip git
# Set working directory
WORKDIR /vllm-installation
RUN ldconfig /usr/local/cuda-$(echo "$WORKER_CUDA_VERSION" | sed 's/\.0$//')/compat/
# Install build and runtime dependencies
COPY vllm/requirements-common.txt requirements-common.txt
COPY vllm/requirements-cuda${WORKER_CUDA_VERSION}.txt requirements-cuda.txt
RUN --mount=type=cache,target=/root/.cache/pip \
pip install -r requirements-cuda.txt
# Install development dependencies
COPY vllm/requirements-dev.txt requirements-dev.txt
RUN --mount=type=cache,target=/root/.cache/pip \
pip install -r requirements-dev.txt
ARG torch_cuda_arch_list='7.0 7.5 8.0 8.6 8.9 9.0+PTX'
ENV TORCH_CUDA_ARCH_LIST=${torch_cuda_arch_list}
FROM dev AS build
# Re-declare ARG after FROM
ARG WORKER_CUDA_VERSION
# Install build dependencies
COPY vllm/requirements-build.txt requirements-build.txt
RUN --mount=type=cache,target=/root/.cache/pip \
pip install -r requirements-build.txt
# install compiler cache to speed up compilation leveraging local or remote caching
RUN apt-get update -y && apt-get install -y ccache
# Copy necessary files
COPY vllm/csrc csrc
COPY vllm/setup.py setup.py
COPY vllm/cmake cmake
COPY vllm/CMakeLists.txt CMakeLists.txt
COPY vllm/requirements-common.txt requirements-common.txt
COPY vllm/requirements-cuda${WORKER_CUDA_VERSION}.txt requirements-cuda.txt
COPY vllm/pyproject.toml pyproject.toml
COPY vllm/vllm vllm
# Set environment variables for building extensions
ENV WORKER_CUDA_VERSION=${WORKER_CUDA_VERSION}
ENV VLLM_INSTALL_PUNICA_KERNELS=0
# Build extensions
ENV CCACHE_DIR=/root/.cache/ccache
RUN --mount=type=cache,target=/root/.cache/ccache \
--mount=type=cache,target=/root/.cache/pip \
python3 setup.py bdist_wheel --dist-dir=dist
RUN --mount=type=cache,target=/root/.cache/pip \
pip cache remove vllm_nccl*
FROM dev as flash-attn-builder
# max jobs used for build
# flash attention version
ARG flash_attn_version=v2.5.8
ENV FLASH_ATTN_VERSION=${flash_attn_version}
WORKDIR /usr/src/flash-attention-v2
# Download the wheel or build it if a pre-compiled release doesn't exist
RUN pip --verbose wheel flash-attn==${FLASH_ATTN_VERSION} \
--no-build-isolation --no-deps --no-cache-dir
FROM dev as NCCL-installer
# Re-declare ARG after FROM
ARG WORKER_CUDA_VERSION
# Update and install necessary libraries
RUN apt-get update -y \
&& apt-get install -y wget
# Install NCCL library
RUN if [ "$WORKER_CUDA_VERSION" = "11.8.0" ]; then \
wget https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.0-1_all.deb \
&& dpkg -i cuda-keyring_1.0-1_all.deb \
&& apt-get update \
&& apt install -y libnccl2=2.15.5-1+cuda11.8 libnccl-dev=2.15.5-1+cuda11.8; \
elif [ "$WORKER_CUDA_VERSION" = "12.1.0" ]; then \
wget https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.0-1_all.deb \
&& dpkg -i cuda-keyring_1.0-1_all.deb \
&& apt-get update \
&& apt install -y libnccl2=2.17.1-1+cuda12.1 libnccl-dev=2.17.1-1+cuda12.1; \
else \
echo "Unsupported CUDA version: $WORKER_CUDA_VERSION"; \
exit 1; \
fi
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-base-ubuntu22.04 AS vllm-base
# Re-declare ARG after FROM
ARG WORKER_CUDA_VERSION
# Update and install necessary libraries
RUN apt-get update -y \
&& apt-get install -y python3-pip
# Set working directory
WORKDIR /vllm-workspace
RUN ldconfig /usr/local/cuda-$(echo "$WORKER_CUDA_VERSION" | sed 's/\.0$//')/compat/
RUN --mount=type=bind,from=build,src=/vllm-installation/dist,target=/vllm-workspace/dist \
--mount=type=cache,target=/root/.cache/pip \
pip install dist/*.whl --verbose
RUN --mount=type=bind,from=flash-attn-builder,src=/usr/src/flash-attention-v2,target=/usr/src/flash-attention-v2 \
--mount=type=cache,target=/root/.cache/pip \
pip install /usr/src/flash-attention-v2/*.whl --no-cache-dir
FROM vllm-base AS runtime
# install additional dependencies for openai api server
RUN --mount=type=cache,target=/root/.cache/pip \
pip install accelerate hf_transfer modelscope tensorizer
# Set PYTHONPATH environment variable
ENV PYTHONPATH="/"
# Copy NCCL library
COPY --from=NCCL-installer /usr/lib/x86_64-linux-gnu/libnccl.so.2 /usr/lib/x86_64-linux-gnu/libnccl.so.2
# Set the VLLM_NCCL_SO_PATH environment variable
ENV VLLM_NCCL_SO_PATH="/usr/lib/x86_64-linux-gnu/libnccl.so.2"
# Validate the installation
RUN python3 -c "import vllm; print(vllm.__file__)"
+2
View File
@@ -0,0 +1,2 @@
version: '0.4.2'
dev_version: '0.4.2'
-109
View File
@@ -1,109 +0,0 @@
################### vLLM Base Dockerfile ###################
# This Dockerfile is for building the image that the
# vLLM worker container will use as its base image.
# If your changes are outside of the vLLM source code, you
# do not need to build this image.
##########################################################
# Define the CUDA version for the build
ARG WORKER_CUDA_VERSION=11.8.0
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-devel-ubuntu22.04 AS dev
# Re-declare ARG after FROM
ARG WORKER_CUDA_VERSION
# Update and install dependencies
RUN apt-get update -y \
&& apt-get install -y python3-pip git
RUN if [ "${WORKER_CUDA_VERSION}" = "12.1.0" ]; then \
ldconfig /usr/local/cuda-12.1/compat/; \
fi
# Set working directory
WORKDIR /vllm-installation
# Install build and runtime dependencies
COPY vllm-${WORKER_CUDA_VERSION}/requirements.txt requirements.txt
RUN --mount=type=cache,target=/root/.cache/pip \
pip install -r requirements.txt
RUN --mount=type=cache,target=/root/.cache/pip \
if [ "${WORKER_CUDA_VERSION}" = "11.8.0" ]; then \
pip install -U --force-reinstall torch==2.1.2 xformers==0.0.23.post1 --index-url https://download.pytorch.org/whl/cu118; \
fi
# Install development dependencies
COPY vllm-${WORKER_CUDA_VERSION}/requirements-dev.txt requirements-dev.txt
RUN --mount=type=cache,target=/root/.cache/pip \
pip install -r requirements-dev.txt
FROM dev AS build
# Re-declare ARG after FROM
ARG WORKER_CUDA_VERSION
# Install build dependencies
COPY vllm-${WORKER_CUDA_VERSION}/requirements-build.txt requirements-build.txt
RUN --mount=type=cache,target=/root/.cache/pip \
pip install -r requirements-build.txt
# Copy necessary files
COPY vllm-${WORKER_CUDA_VERSION}/csrc csrc
COPY vllm-${WORKER_CUDA_VERSION}/setup.py setup.py
COPY vllm-12.1.0/pyproject.toml pyproject.toml
COPY vllm-${WORKER_CUDA_VERSION}/vllm/__init__.py vllm/__init__.py
# Conditional installation based on CUDA version
RUN --mount=type=cache,target=/root/.cache/pip \
if [ "${WORKER_CUDA_VERSION}" = "11.8.0" ]; then \
pip install -U --force-reinstall torch==2.1.2 xformers==0.0.23.post1 --index-url https://download.pytorch.org/whl/cu118; \
rm pyproject.toml; \
elif [ "${WORKER_CUDA_VERSION}" != "12.1.0" ]; then \
echo "WORKER_CUDA_VERSION not supported"; \
exit 1; \
fi
# Set environment variables for building extensions
ARG torch_cuda_arch_list='7.0 7.5 8.0 8.6 8.9 9.0+PTX'
ENV TORCH_CUDA_ARCH_LIST=${torch_cuda_arch_list}
ARG max_jobs=48
ENV MAX_JOBS=${max_jobs}
ARG nvcc_threads=1024
ENV NVCC_THREADS=${nvcc_threads}
# Build extensions
RUN python3 setup.py build_ext --inplace
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-runtime-ubuntu22.04 AS vllm-base
# Re-declare ARG after FROM
ARG WORKER_CUDA_VERSION
# Update and install necessary libraries
RUN apt-get update -y \
&& apt-get install -y python3-pip
# Set working directory
WORKDIR /vllm-installation
# Install runtime dependencies
COPY vllm-${WORKER_CUDA_VERSION}/requirements.txt requirements.txt
RUN --mount=type=cache,target=/root/.cache/pip \
pip install -r requirements.txt
RUN --mount=type=cache,target=/root/.cache/pip \
if [ "${WORKER_CUDA_VERSION}" = "11.8.0" ]; then \
pip install -U --force-reinstall torch==2.1.2 xformers==0.0.23.post1 --index-url https://download.pytorch.org/whl/cu118; \
fi
# Copy built files from the build stage
COPY --from=build /vllm-installation/vllm/*.so /vllm-installation/vllm/
COPY vllm-${WORKER_CUDA_VERSION}/vllm vllm
# Set PYTHONPATH environment variable
ENV PYTHONPATH="/"
# Validate the installation
RUN python3 -c "import sys; print(sys.path); import vllm; print(vllm.__file__)"
-12
View File
@@ -1,12 +0,0 @@
#!/bin/bash
git clone https://github.com/runpod/vllm-fork-for-sls-worker.git
cp -r vllm-fork-for-sls-worker vllm-12.1.0
cp -r vllm-fork-for-sls-worker vllm-11.8.0
rm -rf vllm-fork-for-sls-worker
cd vllm-11.8.0
git checkout cuda-11.8
echo "vLLM Base Image Builder Setup Complete."