Release 0.3.2
This commit is contained in:
@@ -0,0 +1,3 @@
|
|||||||
|
[submodule "vllm-base-image/vllm"]
|
||||||
|
path = vllm-base-image/vllm
|
||||||
|
url = https://github.com/runpod/vllm-fork-for-sls-worker.git
|
||||||
+1
-1
@@ -1,5 +1,5 @@
|
|||||||
ARG WORKER_CUDA_VERSION=11.8.0
|
ARG WORKER_CUDA_VERSION=11.8.0
|
||||||
FROM runpod/worker-vllm:base-0.3.1-cuda${WORKER_CUDA_VERSION} AS vllm-base
|
FROM runpod/worker-vllm:base-0.3.2-cuda${WORKER_CUDA_VERSION} AS vllm-base
|
||||||
|
|
||||||
RUN apt-get update -y \
|
RUN apt-get update -y \
|
||||||
&& apt-get install -y python3-pip
|
&& apt-get install -y python3-pip
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
|
|
||||||
Deploy Blazing-fast LLMs powered by [vLLM](https://github.com/vllm-project/vllm) on RunPod Serverless in a few clicks.
|
Deploy Blazing-fast LLMs powered by [vLLM](https://github.com/vllm-project/vllm) on RunPod Serverless in a few clicks.
|
||||||
|
|
||||||
<p>Worker Version: 0.3.1 | vLLM Version: 0.3.2</p>
|
<p>Worker Version: 0.3.2 | vLLM Version: 0.3.3</p>
|
||||||
|
|
||||||
[](https://github.com/runpod-workers/worker-vllm/actions/workflows/docker-build-release.yml)
|
[](https://github.com/runpod-workers/worker-vllm/actions/workflows/docker-build-release.yml)
|
||||||
|
|
||||||
@@ -88,7 +88,7 @@ This table provides a quick reference to the image tags you should use based on
|
|||||||
**LLM Settings**
|
**LLM Settings**
|
||||||
| `MODEL_NAME`**\*** | - | `str` | Hugging Face Model Repository (e.g., `openchat/openchat-3.5-1210`). |
|
| `MODEL_NAME`**\*** | - | `str` | Hugging Face Model Repository (e.g., `openchat/openchat-3.5-1210`). |
|
||||||
| `MODEL_REVISION` | `None` | `str` |Model revision(branch) to load. |
|
| `MODEL_REVISION` | `None` | `str` |Model revision(branch) to load. |
|
||||||
| `MAX_MODEL_LENGTH` | Model's maximum | `int` |Maximum number of tokens for the engine to handle per request. |
|
| `MAX_MODEL_LEN` | Model's maximum | `int` |Maximum number of tokens for the engine to handle per request. |
|
||||||
| `BASE_PATH` | `/runpod-volume` | `str` |Storage directory for Huggingface cache and model. Utilizes network storage if attached when pointed at `/runpod-volume`, which will have only one worker download the model once, which all workers will be able to load. If no network volume is present, creates a local directory within each worker. |
|
| `BASE_PATH` | `/runpod-volume` | `str` |Storage directory for Huggingface cache and model. Utilizes network storage if attached when pointed at `/runpod-volume`, which will have only one worker download the model once, which all workers will be able to load. If no network volume is present, creates a local directory within each worker. |
|
||||||
| `LOAD_FORMAT` | `auto` | `str` |Format to load model in. |
|
| `LOAD_FORMAT` | `auto` | `str` |Format to load model in. |
|
||||||
| `HF_TOKEN` | - | `str` |Hugging Face token for private and gated models. |
|
| `HF_TOKEN` | - | `str` |Hugging Face token for private and gated models. |
|
||||||
|
|||||||
@@ -45,7 +45,6 @@ if __name__ == "__main__":
|
|||||||
with open("/local_model_path.txt", "w") as f:
|
with open("/local_model_path.txt", "w") as f:
|
||||||
f.write(model_folder)
|
f.write(model_folder)
|
||||||
|
|
||||||
if tokenizer != model:
|
|
||||||
tokenizer_folder = download_extras_or_tokenizer(tokenizer, download_dir, revisions["tokenizer"])
|
tokenizer_folder = download_extras_or_tokenizer(tokenizer, download_dir, revisions["tokenizer"])
|
||||||
with open("/local_tokenizer_path.txt", "w") as f:
|
with open("/local_tokenizer_path.txt", "w") as f:
|
||||||
f.write(tokenizer_folder)
|
f.write(tokenizer_folder)
|
||||||
|
|||||||
@@ -7,3 +7,4 @@ huggingface-hub
|
|||||||
packaging
|
packaging
|
||||||
typing-extensions==4.7.1
|
typing-extensions==4.7.1
|
||||||
pydantic
|
pydantic
|
||||||
|
pydantic-settings
|
||||||
+1
-1
@@ -39,7 +39,7 @@ class EngineConfig:
|
|||||||
"trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))),
|
"trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))),
|
||||||
"gpu_memory_utilization": float(os.getenv("GPU_MEMORY_UTILIZATION", 0.95)),
|
"gpu_memory_utilization": float(os.getenv("GPU_MEMORY_UTILIZATION", 0.95)),
|
||||||
"max_parallel_loading_workers": None if device_count() > 1 or not os.getenv("MAX_PARALLEL_LOADING_WORKERS") else int(os.getenv("MAX_PARALLEL_LOADING_WORKERS")),
|
"max_parallel_loading_workers": None if device_count() > 1 or not os.getenv("MAX_PARALLEL_LOADING_WORKERS") else int(os.getenv("MAX_PARALLEL_LOADING_WORKERS")),
|
||||||
"max_model_len": int(os.getenv("MAX_MODEL_LENGTH")) if os.getenv("MAX_MODEL_LENGTH") else None,
|
"max_model_len": int(os.getenv("MAX_MODEL_LEN")) if os.getenv("MAX_MODEL_LEN") else None,
|
||||||
"tensor_parallel_size": device_count(),
|
"tensor_parallel_size": device_count(),
|
||||||
"seed": int(os.getenv("SEED")) if os.getenv("SEED") else None,
|
"seed": int(os.getenv("SEED")) if os.getenv("SEED") else None,
|
||||||
"kv_cache_dtype": os.getenv("KV_CACHE_DTYPE"),
|
"kv_cache_dtype": os.getenv("KV_CACHE_DTYPE"),
|
||||||
|
|||||||
@@ -1,30 +1,4 @@
|
|||||||
from typing import Union
|
|
||||||
|
|
||||||
DEFAULT_BATCH_SIZE = 50
|
DEFAULT_BATCH_SIZE = 50
|
||||||
DEFAULT_MAX_CONCURRENCY = 300
|
DEFAULT_MAX_CONCURRENCY = 300
|
||||||
DEFAULT_BATCH_SIZE_GROWTH_FACTOR = 3
|
DEFAULT_BATCH_SIZE_GROWTH_FACTOR = 3
|
||||||
DEFAULT_MIN_BATCH_SIZE = 1
|
DEFAULT_MIN_BATCH_SIZE = 1
|
||||||
|
|
||||||
SAMPLING_PARAM_TYPES = {
|
|
||||||
"n": int,
|
|
||||||
"best_of": int,
|
|
||||||
"presence_penalty": float,
|
|
||||||
"frequency_penalty": float,
|
|
||||||
"repetition_penalty": float,
|
|
||||||
"temperature": Union[float, int],
|
|
||||||
"top_p": float,
|
|
||||||
"top_k": int,
|
|
||||||
"min_p": float,
|
|
||||||
"use_beam_search": bool,
|
|
||||||
"length_penalty": float,
|
|
||||||
"early_stopping": Union[bool, str],
|
|
||||||
"stop": Union[str, list],
|
|
||||||
"stop_token_ids": list,
|
|
||||||
"ignore_eos": bool,
|
|
||||||
"max_tokens": int,
|
|
||||||
"logprobs": int,
|
|
||||||
"prompt_logprobs": int,
|
|
||||||
"skip_special_tokens": bool,
|
|
||||||
"spaces_between_special_tokens": bool,
|
|
||||||
"include_stop_str_in_output": bool
|
|
||||||
}
|
|
||||||
+3
-5
@@ -6,7 +6,7 @@ from dotenv import load_dotenv
|
|||||||
from torch.cuda import device_count
|
from torch.cuda import device_count
|
||||||
from typing import AsyncGenerator
|
from typing import AsyncGenerator
|
||||||
|
|
||||||
from vllm import AsyncLLMEngine, AsyncEngineArgs, SamplingParams
|
from vllm import AsyncLLMEngine, AsyncEngineArgs
|
||||||
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
|
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
|
||||||
from vllm.entrypoints.openai.serving_completion import OpenAIServingCompletion
|
from vllm.entrypoints.openai.serving_completion import OpenAIServingCompletion
|
||||||
from vllm.entrypoints.openai.protocol import ChatCompletionRequest, CompletionRequest, ErrorResponse
|
from vllm.entrypoints.openai.protocol import ChatCompletionRequest, CompletionRequest, ErrorResponse
|
||||||
@@ -16,7 +16,6 @@ from constants import DEFAULT_MAX_CONCURRENCY, DEFAULT_BATCH_SIZE, DEFAULT_BATCH
|
|||||||
from tokenizer import TokenizerWrapper
|
from tokenizer import TokenizerWrapper
|
||||||
from config import EngineConfig
|
from config import EngineConfig
|
||||||
|
|
||||||
|
|
||||||
class vLLMEngine:
|
class vLLMEngine:
|
||||||
def __init__(self, engine = None):
|
def __init__(self, engine = None):
|
||||||
load_dotenv() # For local development
|
load_dotenv() # For local development
|
||||||
@@ -35,7 +34,7 @@ class vLLMEngine:
|
|||||||
try:
|
try:
|
||||||
async for batch in self._generate_vllm(
|
async for batch in self._generate_vllm(
|
||||||
llm_input=job_input.llm_input,
|
llm_input=job_input.llm_input,
|
||||||
validated_sampling_params=job_input.validated_sampling_params,
|
validated_sampling_params=job_input.sampling_params,
|
||||||
batch_size=job_input.max_batch_size,
|
batch_size=job_input.max_batch_size,
|
||||||
stream=job_input.stream,
|
stream=job_input.stream,
|
||||||
apply_chat_template=job_input.apply_chat_template,
|
apply_chat_template=job_input.apply_chat_template,
|
||||||
@@ -45,12 +44,11 @@ class vLLMEngine:
|
|||||||
):
|
):
|
||||||
yield batch
|
yield batch
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
yield create_error_response(str(e)).model_dump()
|
yield {"error": create_error_response(str(e)).model_dump()}
|
||||||
|
|
||||||
async def _generate_vllm(self, llm_input, validated_sampling_params, batch_size, stream, apply_chat_template, request_id, batch_size_growth_factor, min_batch_size: str) -> AsyncGenerator[dict, None]:
|
async def _generate_vllm(self, llm_input, validated_sampling_params, batch_size, stream, apply_chat_template, request_id, batch_size_growth_factor, min_batch_size: str) -> AsyncGenerator[dict, None]:
|
||||||
if apply_chat_template or isinstance(llm_input, list):
|
if apply_chat_template or isinstance(llm_input, list):
|
||||||
llm_input = self.tokenizer.apply_chat_template(llm_input)
|
llm_input = self.tokenizer.apply_chat_template(llm_input)
|
||||||
validated_sampling_params = SamplingParams(**validated_sampling_params)
|
|
||||||
results_generator = self.llm.generate(llm_input, validated_sampling_params, request_id)
|
results_generator = self.llm.generate(llm_input, validated_sampling_params, request_id)
|
||||||
n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True
|
n_responses, n_input_tokens, is_first_output = validated_sampling_params.n, 0, True
|
||||||
last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0}
|
last_output_texts, token_counters = ["" for _ in range(n_responses)], {"batch": 0, "total": 0}
|
||||||
|
|||||||
+4
-17
@@ -1,10 +1,9 @@
|
|||||||
import logging
|
import logging
|
||||||
from http import HTTPStatus
|
from http import HTTPStatus
|
||||||
from typing import Any, Dict
|
from typing import Any, Dict
|
||||||
from constants import SAMPLING_PARAM_TYPES
|
|
||||||
from vllm.utils import random_uuid
|
from vllm.utils import random_uuid
|
||||||
from vllm.entrypoints.openai.protocol import ErrorResponse
|
from vllm.entrypoints.openai.protocol import ErrorResponse
|
||||||
|
from vllm import SamplingParams
|
||||||
|
|
||||||
logging.basicConfig(level=logging.INFO)
|
logging.basicConfig(level=logging.INFO)
|
||||||
|
|
||||||
@@ -25,20 +24,6 @@ def count_physical_cores():
|
|||||||
|
|
||||||
return len(cores)
|
return len(cores)
|
||||||
|
|
||||||
def validate_sampling_params(params: Dict[str, Any]) -> Dict[str, Any]:
|
|
||||||
validated_params = {}
|
|
||||||
invalid_params = []
|
|
||||||
for key, value in params.items():
|
|
||||||
expected_type = SAMPLING_PARAM_TYPES.get(key)
|
|
||||||
if expected_type and isinstance(value, expected_type):
|
|
||||||
validated_params[key] = value
|
|
||||||
else:
|
|
||||||
invalid_params.append(key)
|
|
||||||
|
|
||||||
if len(invalid_params) > 0:
|
|
||||||
logging.warning("Ignoring invalid sampling params: %s", invalid_params)
|
|
||||||
|
|
||||||
return validated_params
|
|
||||||
|
|
||||||
class JobInput:
|
class JobInput:
|
||||||
def __init__(self, job):
|
def __init__(self, job):
|
||||||
@@ -47,7 +32,7 @@ class JobInput:
|
|||||||
self.max_batch_size = job.get("max_batch_size")
|
self.max_batch_size = job.get("max_batch_size")
|
||||||
self.apply_chat_template = job.get("apply_chat_template", False)
|
self.apply_chat_template = job.get("apply_chat_template", False)
|
||||||
self.use_openai_format = job.get("use_openai_format", False)
|
self.use_openai_format = job.get("use_openai_format", False)
|
||||||
self.validated_sampling_params = validate_sampling_params(job.get("sampling_params", {}))
|
self.sampling_params = SamplingParams(**job.get("sampling_params", {}))
|
||||||
self.request_id = random_uuid()
|
self.request_id = random_uuid()
|
||||||
batch_size_growth_factor = job.get("batch_size_growth_factor")
|
batch_size_growth_factor = job.get("batch_size_growth_factor")
|
||||||
self.batch_size_growth_factor = float(batch_size_growth_factor) if batch_size_growth_factor else None
|
self.batch_size_growth_factor = float(batch_size_growth_factor) if batch_size_growth_factor else None
|
||||||
@@ -79,3 +64,5 @@ def create_error_response(message: str, err_type: str = "BadRequestError", statu
|
|||||||
return ErrorResponse(message=message,
|
return ErrorResponse(message=message,
|
||||||
type=err_type,
|
type=err_type,
|
||||||
code=status_code.value)
|
code=status_code.value)
|
||||||
|
|
||||||
|
|
||||||
@@ -17,25 +17,16 @@ ARG WORKER_CUDA_VERSION
|
|||||||
RUN apt-get update -y \
|
RUN apt-get update -y \
|
||||||
&& apt-get install -y python3-pip git
|
&& apt-get install -y python3-pip git
|
||||||
|
|
||||||
RUN if [ "${WORKER_CUDA_VERSION}" = "12.1.0" ]; then \
|
|
||||||
ldconfig /usr/local/cuda-12.1/compat/; \
|
|
||||||
fi
|
|
||||||
|
|
||||||
# Set working directory
|
# Set working directory
|
||||||
WORKDIR /vllm-installation
|
WORKDIR /vllm-installation
|
||||||
|
|
||||||
# Install build and runtime dependencies
|
# Install build and runtime dependencies
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/requirements.txt requirements.txt
|
COPY vllm/requirements-${WORKER_CUDA_VERSION}.txt requirements.txt
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
pip install -r requirements.txt
|
pip install -r requirements.txt
|
||||||
|
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
|
||||||
if [ "${WORKER_CUDA_VERSION}" = "11.8.0" ]; then \
|
|
||||||
pip install -U --force-reinstall torch==2.1.2 xformers==0.0.23.post1 --index-url https://download.pytorch.org/whl/cu118; \
|
|
||||||
fi
|
|
||||||
|
|
||||||
# Install development dependencies
|
# Install development dependencies
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/requirements-dev.txt requirements-dev.txt
|
COPY vllm/requirements-dev.txt requirements-dev.txt
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
pip install -r requirements-dev.txt
|
pip install -r requirements-dev.txt
|
||||||
|
|
||||||
@@ -45,25 +36,15 @@ FROM dev AS build
|
|||||||
ARG WORKER_CUDA_VERSION
|
ARG WORKER_CUDA_VERSION
|
||||||
|
|
||||||
# Install build dependencies
|
# Install build dependencies
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/requirements-build.txt requirements-build.txt
|
COPY vllm/requirements-build.txt requirements-build.txt
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
pip install -r requirements-build.txt
|
pip install -r requirements-build.txt
|
||||||
|
|
||||||
# Copy necessary files
|
# Copy necessary files
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/csrc csrc
|
COPY vllm/csrc csrc
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/setup.py setup.py
|
COPY vllm/setup.py setup.py
|
||||||
COPY vllm-12.1.0/pyproject.toml pyproject.toml
|
COPY vllm/pyproject.toml pyproject.toml
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/vllm/__init__.py vllm/__init__.py
|
COPY vllm/vllm/__init__.py vllm/__init__.py
|
||||||
|
|
||||||
# Conditional installation based on CUDA version
|
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
|
||||||
if [ "${WORKER_CUDA_VERSION}" = "11.8.0" ]; then \
|
|
||||||
pip install -U --force-reinstall torch==2.1.2 xformers==0.0.23.post1 --index-url https://download.pytorch.org/whl/cu118; \
|
|
||||||
rm pyproject.toml; \
|
|
||||||
elif [ "${WORKER_CUDA_VERSION}" != "12.1.0" ]; then \
|
|
||||||
echo "WORKER_CUDA_VERSION not supported"; \
|
|
||||||
exit 1; \
|
|
||||||
fi
|
|
||||||
|
|
||||||
# Set environment variables for building extensions
|
# Set environment variables for building extensions
|
||||||
ARG torch_cuda_arch_list='7.0 7.5 8.0 8.6 8.9 9.0+PTX'
|
ARG torch_cuda_arch_list='7.0 7.5 8.0 8.6 8.9 9.0+PTX'
|
||||||
@@ -72,8 +53,10 @@ ARG max_jobs=48
|
|||||||
ENV MAX_JOBS=${max_jobs}
|
ENV MAX_JOBS=${max_jobs}
|
||||||
ARG nvcc_threads=1024
|
ARG nvcc_threads=1024
|
||||||
ENV NVCC_THREADS=${nvcc_threads}
|
ENV NVCC_THREADS=${nvcc_threads}
|
||||||
|
ENV WORKER_CUDA_VERSION=${WORKER_CUDA_VERSION}
|
||||||
|
ENV VLLM_INSTALL_PUNICA_KERNELS=0
|
||||||
# Build extensions
|
# Build extensions
|
||||||
|
RUN ldconfig /usr/local/cuda-$(echo "$WORKER_CUDA_VERSION" | sed 's/\.0$//')/compat/
|
||||||
RUN python3 setup.py build_ext --inplace
|
RUN python3 setup.py build_ext --inplace
|
||||||
|
|
||||||
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-runtime-ubuntu22.04 AS vllm-base
|
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-runtime-ubuntu22.04 AS vllm-base
|
||||||
@@ -88,19 +71,15 @@ RUN apt-get update -y \
|
|||||||
# Set working directory
|
# Set working directory
|
||||||
WORKDIR /vllm-installation
|
WORKDIR /vllm-installation
|
||||||
|
|
||||||
|
|
||||||
# Install runtime dependencies
|
# Install runtime dependencies
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/requirements.txt requirements.txt
|
COPY vllm/requirements-${WORKER_CUDA_VERSION}.txt requirements.txt
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||||
pip install -r requirements.txt
|
pip install -r requirements.txt
|
||||||
|
|
||||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
|
||||||
if [ "${WORKER_CUDA_VERSION}" = "11.8.0" ]; then \
|
|
||||||
pip install -U --force-reinstall torch==2.1.2 xformers==0.0.23.post1 --index-url https://download.pytorch.org/whl/cu118; \
|
|
||||||
fi
|
|
||||||
|
|
||||||
# Copy built files from the build stage
|
# Copy built files from the build stage
|
||||||
COPY --from=build /vllm-installation/vllm/*.so /vllm-installation/vllm/
|
COPY --from=build /vllm-installation/vllm/*.so /vllm-installation/vllm/
|
||||||
COPY vllm-${WORKER_CUDA_VERSION}/vllm vllm
|
COPY vllm/vllm vllm
|
||||||
|
|
||||||
# Set PYTHONPATH environment variable
|
# Set PYTHONPATH environment variable
|
||||||
ENV PYTHONPATH="/"
|
ENV PYTHONPATH="/"
|
||||||
Submodule
+1
Submodule vllm-base-image/vllm added at c46d230a62
@@ -1,12 +0,0 @@
|
|||||||
#!/bin/bash
|
|
||||||
|
|
||||||
git clone https://github.com/runpod/vllm-fork-for-sls-worker.git
|
|
||||||
|
|
||||||
cp -r vllm-fork-for-sls-worker vllm-12.1.0
|
|
||||||
cp -r vllm-fork-for-sls-worker vllm-11.8.0
|
|
||||||
rm -rf vllm-fork-for-sls-worker
|
|
||||||
|
|
||||||
cd vllm-11.8.0
|
|
||||||
git checkout cuda-11.8
|
|
||||||
|
|
||||||
echo "vLLM Base Image Builder Setup Complete."
|
|
||||||
Reference in New Issue
Block a user