Merge pull request #82 from runpod-workers/any-arg-and-refactor

Allow any vLLM engine args as env vars, Update vLLM, refactor
This commit is contained in:
Marut Pandya
2024-08-01 15:28:11 -07:00
committed by GitHub
13 changed files with 404 additions and 325 deletions
+10 -10
View File
@@ -1,16 +1,20 @@
ARG WORKER_CUDA_VERSION=11.8.0
ARG BASE_IMAGE_VERSION=1.0.0
FROM runpod/worker-vllm:base-${BASE_IMAGE_VERSION}-cuda${WORKER_CUDA_VERSION} AS vllm-base
FROM nvidia/cuda:12.1.0-base-ubuntu22.04
RUN apt-get update -y \
&& apt-get install -y python3-pip
RUN ldconfig /usr/local/cuda-12.1/compat/
# Install Python dependencies
COPY builder/requirements.txt /requirements.txt
RUN --mount=type=cache,target=/root/.cache/pip \
python3 -m pip install --upgrade pip && \
python3 -m pip install --upgrade -r /requirements.txt
# Install vLLM (switching back to pip installs since issues that required building fork are fixed and space optimization is not as important since caching) and FlashInfer
RUN python3 -m pip install vllm==0.5.3.post1 && \
python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3
# Setup for Option 2: Building the Image with the Model included
ARG MODEL_NAME=""
ARG TOKENIZER_NAME=""
@@ -32,19 +36,15 @@ ENV MODEL_NAME=$MODEL_NAME \
ENV PYTHONPATH="/:/vllm-workspace"
COPY src/download_model.py /download_model.py
COPY src /src
RUN --mount=type=secret,id=HF_TOKEN,required=false \
if [ -f /run/secrets/HF_TOKEN ]; then \
export HF_TOKEN=$(cat /run/secrets/HF_TOKEN); \
fi && \
if [ -n "$MODEL_NAME" ]; then \
python3 /download_model.py; \
python3 /src/download_model.py; \
fi
# Add source files
COPY src /src
# Remove download_model.py
RUN rm /download_model.py
# Start the handler
CMD ["python3", "/src/handler.py"]
+71 -15
View File
@@ -91,20 +91,76 @@ Below is a summary of the available RunPod Worker images, categorized by image s
#### Environment Variables/Settings
> Note: `0` is equivalent to `False` and `1` is equivalent to `True` for boolean values.
| Name | Default | Type/Choices | Description |
|-------------------------------------|----------------------|-------------------------------------------|-------------|
**LLM Settings**
| `MODEL_NAME`**\*** | - | `str` | Hugging Face Model Repository (e.g., `openchat/openchat-3.5-1210`). |
| `MODEL_REVISION` | `None` | `str` |Model revision(branch) to load. |
| `MAX_MODEL_LEN` | Model's maximum | `int` |Maximum number of tokens for the engine to handle per request. |
| `BASE_PATH` | `/runpod-volume` | `str` |Storage directory for Huggingface cache and model. Utilizes network storage if attached when pointed at `/runpod-volume`, which will have only one worker download the model once, which all workers will be able to load. If no network volume is present, creates a local directory within each worker. |
| `LOAD_FORMAT` | `auto` | `str` |Format to load model in. |
| `HF_TOKEN` | - | `str` |Hugging Face token for private and gated models. |
| `QUANTIZATION` | `None` | `awq`, `squeezellm`, `gptq` |Quantization of given model. The model must already be quantized. |
| `TRUST_REMOTE_CODE` | `0` | boolean as `int` |Trust remote code for Hugging Face models. Can help with Mixtral 8x7B, Quantized models, and unusual models/architectures.
| `SEED` | `0` | `int` |Sets random seed for operations. |
| `KV_CACHE_DTYPE` | `auto` | `auto`, `fp8` |Data type for kv cache storage. Uses `DTYPE` if set to `auto`. |
| `DTYPE` | `auto` | `auto`, `half`, `float16`, `bfloat16`, `float`, `float32` |Sets datatype/precision for model weights and activations. |
| `Name` | `Default` | `Type/Choices` | `Description` |
|-------------------------------------------|-----------------------|--------------------------------------------|---------------|
| `MODEL` | 'facebook/opt-125m' | `str` | Name or path of the Hugging Face model to use. |
| `TOKENIZER` | None | `str` | Name or path of the Hugging Face tokenizer to use. |
| `SKIP_TOKENIZER_INIT` | False | `bool` | Skip initialization of tokenizer and detokenizer. |
| `TOKENIZER_MODE` | 'auto' | ['auto', 'slow'] | The tokenizer mode. |
| `TRUST_REMOTE_CODE` | False | `bool` | Trust remote code from Hugging Face. |
| `DOWNLOAD_DIR` | None | `str` | Directory to download and load the weights. |
| `LOAD_FORMAT` | 'auto' | ['auto', 'pt', 'safetensors', 'npcache', 'dummy', 'tensorizer', 'bitsandbytes'] | The format of the model weights to load. |
| `DTYPE` | 'auto' | ['auto', 'half', 'float16', 'bfloat16', 'float', 'float32'] | Data type for model weights and activations. |
| `KV_CACHE_DTYPE` | 'auto' | ['auto', 'fp8', 'fp8_e5m2', 'fp8_e4m3'] | Data type for KV cache storage. |
| `QUANTIZATION_PARAM_PATH` | None | `str` | Path to the JSON file containing the KV cache scaling factors. |
| `MAX_MODEL_LEN` | None | `int` | Model context length. |
| `GUIDED_DECODING_BACKEND` | 'outlines' | ['outlines', 'lm-format-enforcer'] | Which engine will be used for guided decoding by default. |
| `DISTRIBUTED_EXECUTOR_BACKEND` | None | ['ray', 'mp'] | Backend to use for distributed serving. |
| `WORKER_USE_RAY` | False | `bool` | Deprecated, use --distributed-executor-backend=ray. |
| `PIPELINE_PARALLEL_SIZE` | 1 | `int` | Number of pipeline stages. |
| `TENSOR_PARALLEL_SIZE` | 1 | `int` | Number of tensor parallel replicas. |
| `MAX_PARALLEL_LOADING_WORKERS` | None | `int` | Load model sequentially in multiple batches. |
| `RAY_WORKERS_USE_NSIGHT` | False | `bool` | If specified, use nsight to profile Ray workers. |
| `BLOCK_SIZE` | 16 | [8, 16, 32] | Token block size for contiguous chunks of tokens. |
| `ENABLE_PREFIX_CACHING` | False | `bool` | Enables automatic prefix caching. |
| `DISABLE_SLIDING_WINDOW` | False | `bool` | Disables sliding window, capping to sliding window size. |
| `USE_V2_BLOCK_MANAGER` | False | `bool` | Use BlockSpaceMangerV2. |
| `NUM_LOOKAHEAD_SLOTS` | 0 | `int` | Experimental scheduling config necessary for speculative decoding. |
| `SEED` | 0 | `int` | Random seed for operations. |
| `SWAP_SPACE` | 4 | `int` | CPU swap space size (GiB) per GPU. |
| `GPU_MEMORY_UTILIZATION` | 0.90 | `float` | The fraction of GPU memory to be used for the model executor. |
| `NUM_GPU_BLOCKS_OVERRIDE` | None | `int` | If specified, ignore GPU profiling result and use this number of GPU blocks. |
| `MAX_NUM_BATCHED_TOKENS` | None | `int` | Maximum number of batched tokens per iteration. |
| `MAX_NUM_SEQS` | 256 | `int` | Maximum number of sequences per iteration. |
| `MAX_LOGPROBS` | 20 | `int` | Max number of log probs to return when logprobs is specified in SamplingParams. |
| `DISABLE_LOG_STATS` | False | `bool` | Disable logging statistics. |
| `QUANTIZATION` | None | [*QUANTIZATION_METHODS, None] | Method used to quantize the weights. |
| `ROPE_SCALING` | None | `dict` | RoPE scaling configuration in JSON format. |
| `ROPE_THETA` | None | `float` | RoPE theta. Use with rope_scaling. |
| `ENFORCE_EAGER` | False | `bool` | Always use eager-mode PyTorch. |
| `MAX_CONTEXT_LEN_TO_CAPTURE` | None | `int` | Maximum context length covered by CUDA graphs. |
| `MAX_SEQ_LEN_TO_CAPTURE` | 8192 | `int` | Maximum sequence length covered by CUDA graphs. |
| `DISABLE_CUSTOM_ALL_REDUCE` | False | `bool` | See ParallelConfig. |
| `TOKENIZER_POOL_SIZE` | 0 | `int` | Size of tokenizer pool to use for asynchronous tokenization. |
| `TOKENIZER_POOL_TYPE` | 'ray' | `str` | Type of tokenizer pool to use for asynchronous tokenization. |
| `TOKENIZER_POOL_EXTRA_CONFIG` | None | `dict` | Extra config for tokenizer pool. |
| `ENABLE_LORA` | False | `bool` | If True, enable handling of LoRA adapters. |
| `MAX_LORAS` | 1 | `int` | Max number of LoRAs in a single batch. |
| `MAX_LORA_RANK` | 16 | `int` | Max LoRA rank. |
| `LORA_EXTRA_VOCAB_SIZE` | 256 | `int` | Maximum size of extra vocabulary for LoRA adapters. |
| `LORA_DTYPE` | 'auto' | ['auto', 'float16', 'bfloat16', 'float32'] | Data type for LoRA. |
| `LONG_LORA_SCALING_FACTORS` | None | `tuple` | Specify multiple scaling factors for LoRA adapters. |
| `MAX_CPU_LORAS` | None | `int` | Maximum number of LoRAs to store in CPU memory. |
| `FULLY_SHARDED_LORAS` | False | `bool` | Enable fully sharded LoRA layers. |
| `DEVICE` | 'auto' | ['auto', 'cuda', 'neuron', 'cpu', 'openvino', 'tpu', 'xpu'] | Device type for vLLM execution. |
| `SCHEDULER_DELAY_FACTOR` | 0.0 | `float` | Apply a delay before scheduling next prompt. |
| `ENABLE_CHUNKED_PREFILL` | False | `bool` | Enable chunked prefill requests. |
| `SPECULATIVE_MODEL` | None | `str` | The name of the draft model to be used in speculative decoding. |
| `NUM_SPECULATIVE_TOKENS` | None | `int` | The number of speculative tokens to sample from the draft model. |
| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | None | `int` | Number of tensor parallel replicas for the draft model. |
| `SPECULATIVE_MAX_MODEL_LEN` | None | `int` | The maximum sequence length supported by the draft model. |
| `SPECULATIVE_DISABLE_BY_BATCH_SIZE` | None | `int` | Disable speculative decoding if the number of enqueue requests is larger than this value. |
| `NGRAM_PROMPT_LOOKUP_MAX` | None | `int` | Max size of window for ngram prompt lookup in speculative decoding. |
| `NGRAM_PROMPT_LOOKUP_MIN` | None | `int` | Min size of window for ngram prompt lookup in speculative decoding. |
| `SPEC_DECODING_ACCEPTANCE_METHOD` | 'rejection_sampler' | ['rejection_sampler', 'typical_acceptance_sampler'] | Specify the acceptance method for draft token verification in speculative decoding. |
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD` | None | `float` | Set the lower bound threshold for the posterior probability of a token to be accepted. |
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA` | None | `float` | A scaling factor for the entropy-based threshold for token acceptance. |
| `MODEL_LOADER_EXTRA_CONFIG` | None | `dict` | Extra config for model loader. |
| `PREEMPTION_MODE` | None | `str` | If 'recompute', the engine performs preemption-aware recomputation. If 'save', the engine saves activations into the CPU memory as preemption happens. |
| `PREEMPTION_CHECK_PERIOD` | 1.0 | `float` | How frequently the engine checks if a preemption happens. |
| `PREEMPTION_CPU_CAPACITY` | 2 | `float` | The percentage of CPU memory used for the saved activations. |
| `DISABLE_LOGGING_REQUEST` | False | `bool` | Disable logging requests. |
| `MAX_LOG_LEN` | None | `int` | Max number of prompt characters or prompt ID numbers being printed in log. |
**Tokenizer Settings**
| `TOKENIZER_NAME` | `None` | `str` |Tokenizer repository to use a different tokenizer than the model's default. |
| `TOKENIZER_REVISION` | `None` | `str` |Tokenizer revision to load. |
@@ -149,7 +205,7 @@ To build an image with the model baked in, you must specify the following docker
- `MODEL_REVISION`: Model revision to load (default: `main`).
- `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't. If your intention is to bake the model into the image, you should set this to something like `/models` to make sure there are no issues if you were to accidentally attach network storage.)
- `QUANTIZATION`
- `WORKER_CUDA_VERSION`: `11.8.0` or `12.1.0` (default: `11.8.0` due to a small number of workers not having CUDA 12.1 support yet. `12.1.0` is recommended for optimal performance).
- `WORKER_CUDA_VERSION`: `12.1.0` (`12.1.0` is recommended for optimal performance).
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer)
- `TOKENIZER_REVISION`: Tokenizer revision to load (default: `main`).
+5 -38
View File
@@ -7,54 +7,21 @@ variable "REPOSITORY" {
}
variable "BASE_IMAGE_VERSION" {
default = "1.0.0"
default = "v1.1preview"
}
group "all" {
targets = ["base", "main"]
targets = ["main"]
}
group "base" {
targets = ["base-1180", "base-1210"]
}
group "main" {
targets = ["worker-1180", "worker-1210"]
}
target "base-1180" {
tags = ["${REPOSITORY}/worker-vllm:base-${BASE_IMAGE_VERSION}-cuda11.8.0"]
context = "vllm-base-image"
dockerfile = "Dockerfile"
args = {
WORKER_CUDA_VERSION = "11.8.0"
}
output = ["type=docker,push=${PUSH}"]
}
target "base-1210" {
tags = ["${REPOSITORY}/worker-vllm:base-${BASE_IMAGE_VERSION}-cuda12.1.0"]
context = "vllm-base-image"
dockerfile = "Dockerfile"
args = {
WORKER_CUDA_VERSION = "12.1.0"
}
output = ["type=docker,push=${PUSH}"]
}
target "worker-1180" {
tags = ["${REPOSITORY}/worker-vllm:${BASE_IMAGE_VERSION}-cuda11.8.0"]
context = "."
dockerfile = "Dockerfile"
args = {
BASE_IMAGE_VERSION = "${BASE_IMAGE_VERSION}"
WORKER_CUDA_VERSION = "11.8.0"
}
output = ["type=docker,push=${PUSH}"]
targets = ["worker-1210"]
}
target "worker-1210" {
tags = ["${REPOSITORY}/worker-vllm:${BASE_IMAGE_VERSION}-cuda12.1.0"]
tags = ["${REPOSITORY}/worker-v1-vllm:${BASE_IMAGE_VERSION}-cuda12.1.0"]
context = "."
dockerfile = "Dockerfile"
args = {
-62
View File
@@ -1,62 +0,0 @@
import os
import json
import logging
from dotenv import load_dotenv
from torch.cuda import device_count
from utils import get_int_bool_env
class EngineConfig:
def __init__(self):
load_dotenv()
self.hf_home = os.getenv("HF_HOME")
# Check if /local_metadata.json exists
local_metadata = {}
if os.path.exists("/local_metadata.json"):
with open("/local_metadata.json", "r") as f:
local_metadata = json.load(f)
if local_metadata.get("model_name") is None:
raise ValueError("Model name is not found in /local_metadata.json, there was a problem when you baked the model in.")
logging.info("Using baked-in model")
os.environ["TRANSFORMERS_OFFLINE"] = "1"
os.environ["HF_HUB_OFFLINE"] = "1"
self.model_name_or_path = local_metadata.get("model_name", os.getenv("MODEL_NAME"))
self.model_revision = local_metadata.get("revision", os.getenv("MODEL_REVISION"))
self.tokenizer_name_or_path = local_metadata.get("tokenizer_name", os.getenv("TOKENIZER_NAME")) or self.model_name_or_path
self.tokenizer_revision = local_metadata.get("tokenizer_revision", os.getenv("TOKENIZER_REVISION"))
self.quantization = local_metadata.get("quantization", os.getenv("QUANTIZATION"))
self.config = self._initialize_config()
def _initialize_config(self):
args = {
"model": self.model_name_or_path,
"revision": self.model_revision,
"download_dir": self.hf_home,
"quantization": self.quantization,
"load_format": os.getenv("LOAD_FORMAT", "auto"),
"dtype": os.getenv("DTYPE", "half" if self.quantization else "auto"),
"tokenizer": self.tokenizer_name_or_path,
"tokenizer_revision": self.tokenizer_revision,
"disable_log_stats": get_int_bool_env("DISABLE_LOG_STATS", True),
"disable_log_requests": get_int_bool_env("DISABLE_LOG_REQUESTS", True),
"trust_remote_code": get_int_bool_env("TRUST_REMOTE_CODE", False),
"gpu_memory_utilization": float(os.getenv("GPU_MEMORY_UTILIZATION", 0.95)),
"max_parallel_loading_workers": None if device_count() > 1 or not os.getenv("MAX_PARALLEL_LOADING_WORKERS") else int(os.getenv("MAX_PARALLEL_LOADING_WORKERS")),
"max_model_len": int(os.getenv("MAX_MODEL_LEN")) if os.getenv("MAX_MODEL_LEN") else None,
"tensor_parallel_size": device_count(),
"seed": int(os.getenv("SEED")) if os.getenv("SEED") else None,
"kv_cache_dtype": os.getenv("KV_CACHE_DTYPE"),
"block_size": int(os.getenv("BLOCK_SIZE")) if os.getenv("BLOCK_SIZE") else None,
"swap_space": int(os.getenv("SWAP_SPACE")) if os.getenv("SWAP_SPACE") else None,
"max_seq_len_to_capture": int(os.getenv("MAX_SEQ_LEN_TO_CAPTURE")) if os.getenv("MAX_SEQ_LEN_TO_CAPTURE") else None,
"disable_custom_all_reduce": get_int_bool_env("DISABLE_CUSTOM_ALL_REDUCE", False),
"enforce_eager": get_int_bool_env("ENFORCE_EAGER", False)
}
if args["kv_cache_dtype"] == "fp8_e5m2":
args["kv_cache_dtype"] = "fp8"
logging.warning("Using fp8_e5m2 is deprecated. Please use fp8 instead.")
if os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE"):
args["max_seq_len_to_capture"] = int(os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE"))
logging.warning("Using MAX_CONTEXT_LEN_TO_CAPTURE is deprecated. Please use MAX_SEQ_LEN_TO_CAPTURE instead.")
return {k: v for k, v in args.items() if v not in [None, ""]}
+93 -20
View File
@@ -1,27 +1,100 @@
import os
from huggingface_hub import snapshot_download
import json
import logging
import glob
from shutil import rmtree
from huggingface_hub import snapshot_download
from utils import timer_decorator
BASE_DIR = "/"
TOKENIZER_PATTERNS = [["*.json", "tokenizer*"]]
MODEL_PATTERNS = [["*.safetensors"], ["*.bin"], ["*.pt"]]
def setup_env():
if os.getenv("TESTING_DOWNLOAD") == "1":
BASE_DIR = "tmp"
os.makedirs(BASE_DIR, exist_ok=True)
os.environ.update({
"HF_HOME": f"{BASE_DIR}/hf_cache",
"MODEL_NAME": "openchat/openchat-3.5-0106",
"HF_HUB_ENABLE_HF_TRANSFER": "1",
"TENSORIZE": "1",
"TENSORIZER_NUM_GPUS": "1",
"DTYPE": "auto"
})
@timer_decorator
def download(name, revision, type, cache_dir):
if type == "model":
pattern_sets = [model_pattern + TOKENIZER_PATTERNS[0] for model_pattern in MODEL_PATTERNS]
elif type == "tokenizer":
pattern_sets = TOKENIZER_PATTERNS
else:
raise ValueError(f"Invalid type: {type}")
try:
for pattern_set in pattern_sets:
path = snapshot_download(name, revision=revision, cache_dir=cache_dir,
allow_patterns=pattern_set)
for pattern in pattern_set:
if glob.glob(os.path.join(path, pattern)):
logging.info(f"Successfully downloaded {pattern} model files.")
return path
except ValueError:
raise ValueError(f"No patterns matching {pattern_sets} found for download.")
# @timer_decorator
# def tensorize_model(model_path): TODO: Add back once tensorizer is ready
# from vllm.engine.arg_utils import EngineArgs
# from vllm.model_executor.model_loader.tensorizer import TensorizerConfig, tensorize_vllm_model
# from torch.cuda import device_count
# tensorizer_num_gpus = int(os.getenv("TENSORIZER_NUM_GPUS", "1"))
# if tensorizer_num_gpus > device_count():
# raise ValueError(f"TENSORIZER_NUM_GPUS ({tensorizer_num_gpus}) exceeds available GPUs ({device_count()})")
# dtype = os.getenv("DTYPE", "auto")
# serialized_dir = f"{BASE_DIR}/serialized_model"
# os.makedirs(serialized_dir, exist_ok=True)
# serialized_uri = f"{serialized_dir}/model{'-%03d' if tensorizer_num_gpus > 1 else ''}.tensors"
# tensorize_vllm_model(
# EngineArgs(model=model_path, tensor_parallel_size=tensorizer_num_gpus, dtype=dtype),
# TensorizerConfig(tensorizer_uri=serialized_uri)
# )
# logging.info("Successfully serialized model to %s", str(serialized_uri))
# logging.info("Removing HF Model files after serialization")
# rmtree("/".join(model_path.split("/")[:-2]))
# return serialized_uri, tensorizer_num_gpus, dtype
if __name__ == "__main__":
model_name = os.getenv("MODEL_NAME")
if not model_name:
raise ValueError("Must specify model name by adding --build-arg MODEL_NAME=<your model's repo>")
revision = os.getenv("MODEL_REVISION") or None
snapshot_download(model_name, revision=revision, cache_dir=os.getenv("HF_HOME"))
setup_env()
cache_dir = os.getenv("HF_HOME")
model_name, model_revision = os.getenv("MODEL_NAME"), os.getenv("MODEL_REVISION") or None
tokenizer_name, tokenizer_revision = os.getenv("TOKENIZER_NAME") or model_name, os.getenv("TOKENIZER_REVISION") or model_revision
model_path = download(model_name, model_revision, "model", cache_dir)
metadata = {
"MODEL_NAME": model_path,
"MODEL_REVISION": os.getenv("MODEL_REVISION"),
"QUANTIZATION": os.getenv("QUANTIZATION"),
}
tokenizer_name = os.getenv("TOKENIZER_NAME") or None
tokenizer_revision = os.getenv("TOKENIZER_REVISION") or None
if tokenizer_name:
snapshot_download(tokenizer_name, revision=tokenizer_revision, cache_dir=os.getenv("HF_HOME"))
# if os.getenv("TENSORIZE") == "1": TODO: Add back once tensorizer is ready
# serialized_uri, tensorizer_num_gpus, dtype = tensorize_model(model_path)
# metadata.update({
# "MODEL_NAME": serialized_uri,
# "TENSORIZER_URI": serialized_uri,
# "TENSOR_PARALLEL_SIZE": tensorizer_num_gpus,
# "DTYPE": dtype
# })
# Create file with metadata of baked in model and/or tokenizer
tokenizer_path = download(tokenizer_name, tokenizer_revision, "tokenizer", cache_dir)
metadata.update({
"TOKENIZER_NAME": tokenizer_path,
"TOKENIZER_REVISION": tokenizer_revision
})
with open("/local_metadata.json", "w") as f:
json.dump({
"model_name": model_name,
"revision": revision,
"tokenizer_name": tokenizer_name or model_name,
"tokenizer_revision": tokenizer_revision or revision,
"quantization": os.getenv("QUANTIZATION")
}, f)
with open(f"{BASE_DIR}/local_model_args.json", "w") as f:
json.dump({k: v for k, v in metadata.items() if v not in (None, "")}, f)
+34 -20
View File
@@ -1,13 +1,13 @@
import os
import logging
import json
import asyncio
from dotenv import load_dotenv
from torch.cuda import device_count
from typing import AsyncGenerator
import time
from vllm import AsyncLLMEngine, AsyncEngineArgs
from vllm import AsyncLLMEngine
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
from vllm.entrypoints.openai.serving_completion import OpenAIServingCompletion
from vllm.entrypoints.openai.protocol import ChatCompletionRequest, CompletionRequest, ErrorResponse
@@ -15,14 +15,17 @@ from vllm.entrypoints.openai.protocol import ChatCompletionRequest, CompletionRe
from utils import DummyRequest, JobInput, BatchSize, create_error_response
from constants import DEFAULT_MAX_CONCURRENCY, DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MIN_BATCH_SIZE
from tokenizer import TokenizerWrapper
from config import EngineConfig
from engine_args import get_engine_args
class vLLMEngine:
def __init__(self, engine = None):
load_dotenv() # For local development
self.config = EngineConfig().config
self.tokenizer = TokenizerWrapper(self.config.get("tokenizer"), self.config.get("tokenizer_revision"), self.config.get("trust_remote_code"))
self.llm = self._initialize_llm() if engine is None else engine
self.engine_args = get_engine_args()
logging.info(f"Engine args: {self.engine_args}")
self.tokenizer = TokenizerWrapper(self.engine_args.tokenizer or self.engine_args.model,
self.engine_args.tokenizer_revision,
self.engine_args.trust_remote_code)
self.llm = self._initialize_llm() if engine is None else engine.llm
self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
self.default_batch_size = int(os.getenv("DEFAULT_BATCH_SIZE", DEFAULT_BATCH_SIZE))
self.batch_size_growth_factor = int(os.getenv("BATCH_SIZE_GROWTH_FACTOR", DEFAULT_BATCH_SIZE_GROWTH_FACTOR))
@@ -102,7 +105,7 @@ class vLLMEngine:
def _initialize_llm(self):
try:
start = time.time()
engine = AsyncLLMEngine.from_engine_args(AsyncEngineArgs(**self.config))
engine = AsyncLLMEngine.from_engine_args(self.engine_args)
end = time.time()
logging.info(f"Initialized vLLM engine in {end - start:.2f}s")
return engine
@@ -111,24 +114,35 @@ class vLLMEngine:
raise e
class OpenAIvLLMEngine:
class OpenAIvLLMEngine(vLLMEngine):
def __init__(self, vllm_engine):
self.config = vllm_engine.config
self.llm = vllm_engine.llm
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.config["model"]
super().__init__(vllm_engine)
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.engine_args.model
self.response_role = os.getenv("OPENAI_RESPONSE_ROLE") or "assistant"
self.tokenizer = vllm_engine.tokenizer
self.default_batch_size = vllm_engine.default_batch_size
self.batch_size_growth_factor, self.min_batch_size = vllm_engine.batch_size_growth_factor, vllm_engine.min_batch_size
self._initialize_engines()
asyncio.run(self._initialize_engines())
self.raw_openai_output = bool(int(os.getenv("RAW_OPENAI_OUTPUT", 1)))
def _initialize_engines(self):
async def _initialize_engines(self):
self.model_config = await self.llm.get_model_config()
self.chat_engine = OpenAIServingChat(
self.llm, self.served_model_name, self.response_role,
chat_template=self.tokenizer.tokenizer.chat_template
engine=self.llm,
model_config=self.model_config,
served_model_names=[self.served_model_name],
response_role=self.response_role,
chat_template=self.tokenizer.tokenizer.chat_template,
lora_modules=None,
prompt_adapters=None,
request_logger=None
)
self.completion_engine = OpenAIServingCompletion(
engine=self.llm,
model_config=self.model_config,
served_model_names=[self.served_model_name],
lora_modules=[],
prompt_adapters=None,
request_logger=None
)
self.completion_engine = OpenAIServingCompletion(self.llm, self.served_model_name)
async def generate(self, openai_request: JobInput):
if openai_request.openai_route == "/v1/models":
@@ -162,7 +176,7 @@ class OpenAIvLLMEngine:
yield create_error_response(str(e)).model_dump()
return
response_generator = await generator_function(request, DummyRequest())
response_generator = await generator_function(request, raw_request=None)
if not openai_request.openai_input.get("stream") or isinstance(response_generator, ErrorResponse):
yield response_generator.model_dump()
+169
View File
@@ -0,0 +1,169 @@
import os
import json
import logging
from torch.cuda import device_count
from vllm import AsyncEngineArgs
from vllm.model_executor.model_loader.tensorizer import TensorizerConfig
RENAME_ARGS_MAP = {
"MODEL_NAME": "model",
"MODEL_REVISION": "revision",
"TOKENIZER_NAME": "tokenizer",
"MAX_CONTEXT_LEN_TO_CAPTURE": "max_seq_len_to_capture"
}
DEFAULT_ARGS = {
"disable_log_stats": True,
"disable_log_requests": True,
"gpu_memory_utilization": 0.9,
"pipeline_parallel_size": int(os.getenv('PIPELINE_PARALLEL_SIZE', 1)),
"tensor_parallel_size": int(os.getenv('TENSOR_PARALLEL_SIZE', 1)),
"served_model_name": os.getenv('SERVED_MODEL_NAME', None),
"tokenizer": os.getenv('TOKENIZER', None),
"skip_tokenizer_init": os.getenv('SKIP_TOKENIZER_INIT', 'False').lower() == 'true',
"tokenizer_mode": os.getenv('TOKENIZER_MODE', 'auto'),
"trust_remote_code": os.getenv('TRUST_REMOTE_CODE', 'False').lower() == 'true',
"download_dir": os.getenv('DOWNLOAD_DIR', None),
"load_format": os.getenv('LOAD_FORMAT', 'auto'),
"dtype": os.getenv('DTYPE', 'auto'),
"kv_cache_dtype": os.getenv('KV_CACHE_DTYPE', 'auto'),
"quantization_param_path": os.getenv('QUANTIZATION_PARAM_PATH', None),
"seed": int(os.getenv('SEED', 0)),
"max_model_len": int(os.getenv('MAX_MODEL_LEN', 0)) or None,
"worker_use_ray": os.getenv('WORKER_USE_RAY', 'False').lower() == 'true',
"distributed_executor_backend": os.getenv('DISTRIBUTED_EXECUTOR_BACKEND', None),
"max_parallel_loading_workers": int(os.getenv('MAX_PARALLEL_LOADING_WORKERS', 0)) or None,
"block_size": int(os.getenv('BLOCK_SIZE', 16)),
"enable_prefix_caching": os.getenv('ENABLE_PREFIX_CACHING', 'False').lower() == 'true',
"disable_sliding_window": os.getenv('DISABLE_SLIDING_WINDOW', 'False').lower() == 'true',
"use_v2_block_manager": os.getenv('USE_V2_BLOCK_MANAGER', 'False').lower() == 'true',
"swap_space": int(os.getenv('SWAP_SPACE', 4)), # GiB
"cpu_offload_gb": int(os.getenv('CPU_OFFLOAD_GB', 0)), # GiB
"max_num_batched_tokens": int(os.getenv('MAX_NUM_BATCHED_TOKENS', 0)) or None,
"max_num_seqs": int(os.getenv('MAX_NUM_SEQS', 256)),
"max_logprobs": int(os.getenv('MAX_LOGPROBS', 20)), # Default value for OpenAI Chat Completions API
"revision": os.getenv('REVISION', None),
"code_revision": os.getenv('CODE_REVISION', None),
"rope_scaling": os.getenv('ROPE_SCALING', None),
"rope_theta": float(os.getenv('ROPE_THETA', 0)) or None,
"tokenizer_revision": os.getenv('TOKENIZER_REVISION', None),
"quantization": os.getenv('QUANTIZATION', None),
"enforce_eager": os.getenv('ENFORCE_EAGER', 'False').lower() == 'true',
"max_context_len_to_capture": int(os.getenv('MAX_CONTEXT_LEN_TO_CAPTURE', 0)) or None,
"max_seq_len_to_capture": int(os.getenv('MAX_SEQ_LEN_TO_CAPTURE', 8192)),
"disable_custom_all_reduce": os.getenv('DISABLE_CUSTOM_ALL_REDUCE', 'False').lower() == 'true',
"tokenizer_pool_size": int(os.getenv('TOKENIZER_POOL_SIZE', 0)),
"tokenizer_pool_type": os.getenv('TOKENIZER_POOL_TYPE', 'ray'),
"tokenizer_pool_extra_config": os.getenv('TOKENIZER_POOL_EXTRA_CONFIG', None),
"enable_lora": os.getenv('ENABLE_LORA', 'False').lower() == 'true',
"max_loras": int(os.getenv('MAX_LORAS', 1)),
"max_lora_rank": int(os.getenv('MAX_LORA_RANK', 16)),
"enable_prompt_adapter": os.getenv('ENABLE_PROMPT_ADAPTER', 'False').lower() == 'true',
"max_prompt_adapters": int(os.getenv('MAX_PROMPT_ADAPTERS', 1)),
"max_prompt_adapter_token": int(os.getenv('MAX_PROMPT_ADAPTER_TOKEN', 0)),
"fully_sharded_loras": os.getenv('FULLY_SHARDED_LORAS', 'False').lower() == 'true',
"lora_extra_vocab_size": int(os.getenv('LORA_EXTRA_VOCAB_SIZE', 256)),
"long_lora_scaling_factors": tuple(map(float, os.getenv('LONG_LORA_SCALING_FACTORS', '').split(','))) if os.getenv('LONG_LORA_SCALING_FACTORS') else None,
"lora_dtype": os.getenv('LORA_DTYPE', 'auto'),
"max_cpu_loras": int(os.getenv('MAX_CPU_LORAS', 0)) or None,
"device": os.getenv('DEVICE', 'auto'),
"ray_workers_use_nsight": os.getenv('RAY_WORKERS_USE_NSIGHT', 'False').lower() == 'true',
"num_gpu_blocks_override": int(os.getenv('NUM_GPU_BLOCKS_OVERRIDE', 0)) or None,
"num_lookahead_slots": int(os.getenv('NUM_LOOKAHEAD_SLOTS', 0)),
"model_loader_extra_config": os.getenv('MODEL_LOADER_EXTRA_CONFIG', None),
"ignore_patterns": os.getenv('IGNORE_PATTERNS', None),
"preemption_mode": os.getenv('PREEMPTION_MODE', None),
"scheduler_delay_factor": float(os.getenv('SCHEDULER_DELAY_FACTOR', 0.0)),
"enable_chunked_prefill": os.getenv('ENABLE_CHUNKED_PREFILL', None),
"guided_decoding_backend": os.getenv('GUIDED_DECODING_BACKEND', 'outlines'),
"speculative_model": os.getenv('SPECULATIVE_MODEL', None),
"speculative_draft_tensor_parallel_size": int(os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE', 0)) or None,
"num_speculative_tokens": int(os.getenv('NUM_SPECULATIVE_TOKENS', 0)) or None,
"speculative_max_model_len": int(os.getenv('SPECULATIVE_MAX_MODEL_LEN', 0)) or None,
"speculative_disable_by_batch_size": int(os.getenv('SPECULATIVE_DISABLE_BY_BATCH_SIZE', 0)) or None,
"ngram_prompt_lookup_max": int(os.getenv('NGRAM_PROMPT_LOOKUP_MAX', 0)) or None,
"ngram_prompt_lookup_min": int(os.getenv('NGRAM_PROMPT_LOOKUP_MIN', 0)) or None,
"spec_decoding_acceptance_method": os.getenv('SPEC_DECODING_ACCEPTANCE_METHOD', 'rejection_sampler'),
"typical_acceptance_sampler_posterior_threshold": float(os.getenv('TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD', 0)) or None,
"typical_acceptance_sampler_posterior_alpha": float(os.getenv('TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA', 0)) or None,
"qlora_adapter_name_or_path": os.getenv('QLORA_ADAPTER_NAME_OR_PATH', None),
"disable_logprobs_during_spec_decoding": os.getenv('DISABLE_LOGPROBS_DURING_SPEC_DECODING', None),
"otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None)
}
def match_vllm_args(args):
"""Rename args to match vllm by:
1. Renaming keys to lower case
2. Renaming keys to match vllm
3. Filtering args to match vllm's AsyncEngineArgs
Args:
args (dict): Dictionary of args
Returns:
dict: Dictionary of args with renamed keys
"""
renamed_args = {RENAME_ARGS_MAP.get(k, k): v for k, v in args.items()}
matched_args = {k: v for k, v in renamed_args.items() if k in AsyncEngineArgs.__dataclass_fields__}
return {k: v for k, v in matched_args.items() if v not in [None, ""]}
def get_local_args():
"""
Retrieve local arguments from a JSON file.
Returns:
dict: Local arguments.
"""
if not os.path.exists("/local_model_args.json"):
return {}
with open("/local_model_args.json", "r") as f:
local_args = json.load(f)
if local_args.get("MODEL_NAME") is None:
raise ValueError("Model name not found in /local_model_args.json. There was a problem when baking the model in.")
logging.info(f"Using baked in model with args: {local_args}")
os.environ["TRANSFORMERS_OFFLINE"] = "1"
os.environ["HF_HUB_OFFLINE"] = "1"
return local_args
def get_engine_args():
# Start with default args
args = DEFAULT_ARGS
# Get env args that match keys in AsyncEngineArgs
args.update(os.environ)
# Get local args if model is baked in and overwrite env args
args.update(get_local_args())
# if args.get("TENSORIZER_URI"): TODO: add back once tensorizer is ready
# args["load_format"] = "tensorizer"
# args["model_loader_extra_config"] = TensorizerConfig(tensorizer_uri=args["TENSORIZER_URI"], num_readers=None)
# logging.info(f"Using tensorized model from {args['TENSORIZER_URI']}")
# Rename and match to vllm args
args = match_vllm_args(args)
# Set tensor parallel size and max parallel loading workers if more than 1 GPU is available
num_gpus = device_count()
if num_gpus > 1:
args["tensor_parallel_size"] = num_gpus
args["max_parallel_loading_workers"] = None
if os.getenv("MAX_PARALLEL_LOADING_WORKERS"):
logging.warning("Overriding MAX_PARALLEL_LOADING_WORKERS with None because more than 1 GPU is available.")
# Deprecated env args backwards compatibility
if args.get("kv_cache_dtype") == "fp8_e5m2":
args["kv_cache_dtype"] = "fp8"
logging.warning("Using fp8_e5m2 is deprecated. Please use fp8 instead.")
if os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE"):
args["max_seq_len_to_capture"] = int(os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE"))
logging.warning("Using MAX_CONTEXT_LEN_TO_CAPTURE is deprecated. Please use MAX_SEQ_LEN_TO_CAPTURE instead.")
if "gemma-2" in args.get("model", "").lower():
os.environ["VLLM_ATTENTION_BACKEND"] = "FLASHINFER"
logging.info("Using FLASHINFER for gemma-2 model.")
return AsyncEngineArgs(**args)
+2 -1
View File
@@ -4,7 +4,8 @@ from typing import Union
class TokenizerWrapper:
def __init__(self, tokenizer_name_or_path, tokenizer_revision, trust_remote_code):
self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_name_or_path, revision=tokenizer_revision, trust_remote_code=trust_remote_code)
print(f"tokenizer_name_or_path: {tokenizer_name_or_path}, tokenizer_revision: {tokenizer_revision}, trust_remote_code: {trust_remote_code}")
self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_name_or_path, revision=tokenizer_revision or "main", trust_remote_code=trust_remote_code)
self.custom_chat_template = os.getenv("CUSTOM_CHAT_TEMPLATE")
self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template)
if self.custom_chat_template and isinstance(self.custom_chat_template, str):
+19 -6
View File
@@ -1,9 +1,16 @@
import os
import logging
from http import HTTPStatus
from vllm.utils import random_uuid
from vllm.entrypoints.openai.protocol import ErrorResponse
from vllm import SamplingParams
from functools import wraps
from time import time
try:
from vllm.utils import random_uuid
from vllm.entrypoints.openai.protocol import ErrorResponse
from vllm import SamplingParams
except ImportError:
logging.warning("Error importing vllm, skipping related imports. This is ONLY expected when baking model into docker image from a machine without GPUs")
pass
logging.basicConfig(level=logging.INFO)
@@ -68,6 +75,12 @@ def create_error_response(message: str, err_type: str = "BadRequestError", statu
def get_int_bool_env(env_var: str, default: bool) -> bool:
return int(os.getenv(env_var, int(default))) == 1
def timer_decorator(func):
@wraps(func)
def wrapper(*args, **kwargs):
start = time()
result = func(*args, **kwargs)
end = time()
logging.info(f"{func.__name__} completed in {end - start:.2f} seconds")
return result
return wrapper
-149
View File
@@ -1,149 +0,0 @@
################### vLLM Base Dockerfile ###################
# This Dockerfile is for building the image that the
# vLLM worker container will use as its base image.
# If your changes are outside of the vLLM source code, you
# do not need to build this image.
##########################################################
# Define the CUDA version for the build
ARG WORKER_CUDA_VERSION=11.8.0
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-devel-ubuntu22.04 AS dev
# Re-declare ARG after FROM
ARG WORKER_CUDA_VERSION
# Update and install dependencies
RUN apt-get update -y \
&& apt-get install -y python3-pip git
# Set working directory
WORKDIR /vllm-installation
RUN ldconfig /usr/local/cuda-$(echo "$WORKER_CUDA_VERSION" | sed 's/\.0$//')/compat/
# Install build and runtime dependencies
COPY vllm/requirements-common.txt requirements-common.txt
COPY vllm/requirements-cuda${WORKER_CUDA_VERSION}.txt requirements-cuda.txt
RUN --mount=type=cache,target=/root/.cache/pip \
pip install -r requirements-cuda.txt
# Install development dependencies
COPY vllm/requirements-dev.txt requirements-dev.txt
RUN --mount=type=cache,target=/root/.cache/pip \
pip install -r requirements-dev.txt
ARG torch_cuda_arch_list='7.0 7.5 8.0 8.6 8.9 9.0+PTX'
ENV TORCH_CUDA_ARCH_LIST=${torch_cuda_arch_list}
FROM dev AS build
# Re-declare ARG after FROM
ARG WORKER_CUDA_VERSION
# Install build dependencies
COPY vllm/requirements-build.txt requirements-build.txt
RUN --mount=type=cache,target=/root/.cache/pip \
pip install -r requirements-build.txt
# install compiler cache to speed up compilation leveraging local or remote caching
RUN apt-get update -y && apt-get install -y ccache
# Copy necessary files
COPY vllm/csrc csrc
COPY vllm/setup.py setup.py
COPY vllm/cmake cmake
COPY vllm/CMakeLists.txt CMakeLists.txt
COPY vllm/requirements-common.txt requirements-common.txt
COPY vllm/requirements-cuda${WORKER_CUDA_VERSION}.txt requirements-cuda.txt
COPY vllm/pyproject.toml pyproject.toml
COPY vllm/vllm vllm
# Set environment variables for building extensions
ENV WORKER_CUDA_VERSION=${WORKER_CUDA_VERSION}
ENV VLLM_INSTALL_PUNICA_KERNELS=0
# Build extensions
ENV CCACHE_DIR=/root/.cache/ccache
RUN --mount=type=cache,target=/root/.cache/ccache \
--mount=type=cache,target=/root/.cache/pip \
python3 setup.py bdist_wheel --dist-dir=dist
RUN --mount=type=cache,target=/root/.cache/pip \
pip cache remove vllm_nccl*
FROM dev as flash-attn-builder
# max jobs used for build
# flash attention version
ARG flash_attn_version=v2.5.8
ENV FLASH_ATTN_VERSION=${flash_attn_version}
WORKDIR /usr/src/flash-attention-v2
# Download the wheel or build it if a pre-compiled release doesn't exist
RUN pip --verbose wheel flash-attn==${FLASH_ATTN_VERSION} \
--no-build-isolation --no-deps --no-cache-dir
FROM dev as NCCL-installer
# Re-declare ARG after FROM
ARG WORKER_CUDA_VERSION
# Update and install necessary libraries
RUN apt-get update -y \
&& apt-get install -y wget
# Install NCCL library
RUN if [ "$WORKER_CUDA_VERSION" = "11.8.0" ]; then \
wget https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.0-1_all.deb \
&& dpkg -i cuda-keyring_1.0-1_all.deb \
&& apt-get update \
&& apt install -y libnccl2=2.15.5-1+cuda11.8 libnccl-dev=2.15.5-1+cuda11.8; \
elif [ "$WORKER_CUDA_VERSION" = "12.1.0" ]; then \
wget https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2204/x86_64/cuda-keyring_1.0-1_all.deb \
&& dpkg -i cuda-keyring_1.0-1_all.deb \
&& apt-get update \
&& apt install -y libnccl2=2.17.1-1+cuda12.1 libnccl-dev=2.17.1-1+cuda12.1; \
else \
echo "Unsupported CUDA version: $WORKER_CUDA_VERSION"; \
exit 1; \
fi
FROM nvidia/cuda:${WORKER_CUDA_VERSION}-base-ubuntu22.04 AS vllm-base
# Re-declare ARG after FROM
ARG WORKER_CUDA_VERSION
# Update and install necessary libraries
RUN apt-get update -y \
&& apt-get install -y python3-pip
# Set working directory
WORKDIR /vllm-workspace
RUN ldconfig /usr/local/cuda-$(echo "$WORKER_CUDA_VERSION" | sed 's/\.0$//')/compat/
RUN --mount=type=bind,from=build,src=/vllm-installation/dist,target=/vllm-workspace/dist \
--mount=type=cache,target=/root/.cache/pip \
pip install dist/*.whl --verbose
RUN --mount=type=bind,from=flash-attn-builder,src=/usr/src/flash-attention-v2,target=/usr/src/flash-attention-v2 \
--mount=type=cache,target=/root/.cache/pip \
pip install /usr/src/flash-attention-v2/*.whl --no-cache-dir
FROM vllm-base AS runtime
# install additional dependencies for openai api server
RUN --mount=type=cache,target=/root/.cache/pip \
pip install accelerate hf_transfer modelscope tensorizer
# Set PYTHONPATH environment variable
ENV PYTHONPATH="/"
# Copy NCCL library
COPY --from=NCCL-installer /usr/lib/x86_64-linux-gnu/libnccl.so.2 /usr/lib/x86_64-linux-gnu/libnccl.so.2
# Set the VLLM_NCCL_SO_PATH environment variable
ENV VLLM_NCCL_SO_PATH="/usr/lib/x86_64-linux-gnu/libnccl.so.2"
# Validate the installation
RUN python3 -c "import vllm; print(vllm.__file__)"
-1
View File
@@ -1 +0,0 @@
This directory is for building the vllm-base image utilized by the worker.
-2
View File
@@ -1,2 +0,0 @@
version: '0.4.2'
dev_version: '0.4.2'