Fix Model and Tokenizer download for bake-in option, add revision configuration for both.

This commit is contained in:
Alpay Ariyak
2024-02-02 19:58:50 -05:00
committed by GitHub
5 changed files with 97 additions and 46 deletions
+12 -4
View File
@@ -10,15 +10,18 @@ RUN --mount=type=cache,target=/root/.cache/pip \
python3 -m pip install --upgrade pip && \ python3 -m pip install --upgrade pip && \
python3 -m pip install --upgrade -r /requirements.txt python3 -m pip install --upgrade -r /requirements.txt
# Add source files
COPY src /src
# Setup for Option 2: Building the Image with the Model included # Setup for Option 2: Building the Image with the Model included
ARG MODEL_NAME="" ARG MODEL_NAME=""
ARG TOKENIZER_NAME=""
ARG BASE_PATH="/runpod-volume" ARG BASE_PATH="/runpod-volume"
ARG QUANTIZATION="" ARG QUANTIZATION=""
ARG MODEL_REVISION=""
ARG TOKENIZER_REVISION=""
ENV MODEL_NAME=$MODEL_NAME \ ENV MODEL_NAME=$MODEL_NAME \
MODEL_REVISION=$REVISION \
TOKENIZER_NAME=$TOKENIZER_NAME \
TOKENIZER_REVISION=$TOKENIZER_REVISION \
BASE_PATH=$BASE_PATH \ BASE_PATH=$BASE_PATH \
QUANTIZATION=$QUANTIZATION \ QUANTIZATION=$QUANTIZATION \
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \ HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
@@ -28,13 +31,18 @@ ENV MODEL_NAME=$MODEL_NAME \
ENV PYTHONPATH="/:/vllm-installation" ENV PYTHONPATH="/:/vllm-installation"
COPY builder/download_model.py /download_model.py
RUN --mount=type=secret,id=HF_TOKEN,required=false \ RUN --mount=type=secret,id=HF_TOKEN,required=false \
if [ -f /run/secrets/HF_TOKEN ]; then \ if [ -f /run/secrets/HF_TOKEN ]; then \
export HF_TOKEN=$(cat /run/secrets/HF_TOKEN); \ export HF_TOKEN=$(cat /run/secrets/HF_TOKEN); \
fi && \ fi && \
if [ -n "$MODEL_NAME" ]; then \ if [ -n "$MODEL_NAME" ]; then \
python3 /src/download_model.py; \ python3 /download_model.py; \
fi fi
# Add source files
COPY src /src
# Start the handler # Start the handler
CMD ["python3", "/src/handler.py"] CMD ["python3", "/src/handler.py"]
+10 -4
View File
@@ -58,8 +58,7 @@ Development Image: ```runpod/worker-vllm:dev```
**Optional**: **Optional**:
- LLM Settings: - LLM Settings:
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`) - `MODEL_REVISION`: Model revision to load (default: `None`).
- `CUSTOM_CHAT_TEMPLATE`: Custom chat jinja template, read more about Hugging Face chat templates [here](https://huggingface.co/docs/transformers/chat_templating). (default: `None`)
- `MAX_MODEL_LENGTH`: Maximum number of tokens for the engine to be able to handle. (default: maximum supported by the model) - `MAX_MODEL_LENGTH`: Maximum number of tokens for the engine to be able to handle. (default: maximum supported by the model)
- `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't) - `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't)
- `LOAD_FORMAT`: Format to load model in (default: `auto`). - `LOAD_FORMAT`: Format to load model in (default: `auto`).
@@ -67,15 +66,19 @@ Development Image: ```runpod/worker-vllm:dev```
- `QUANTIZATION`: AWQ (`awq`), SqueezeLLM (`squeezellm`) or GPTQ (`gptq`) Quantization. The specified Model Repo must be of a quantized model. (default: `None`) - `QUANTIZATION`: AWQ (`awq`), SqueezeLLM (`squeezellm`) or GPTQ (`gptq`) Quantization. The specified Model Repo must be of a quantized model. (default: `None`)
- `TRUST_REMOTE_CODE`: Trust remote code for Hugging Face (default: `0`) - `TRUST_REMOTE_CODE`: Trust remote code for Hugging Face (default: `0`)
- Tensor Parallelism: - Tokenizer Settings:
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer)
- `TOKENIZER_REVISION`: Tokenizer revision to load (default: `None`).
- `CUSTOM_CHAT_TEMPLATE`: Custom chat jinja template, read more about Hugging Face chat templates [here](https://huggingface.co/docs/transformers/chat_templating). (default: `None`)
- Tensor Parallelism:
Note that the more GPUs you split a model's weights accross, the slower it will be due to inter-GPU communication overhead. If you can fit the model on a single GPU, it is recommended to do so. Note that the more GPUs you split a model's weights accross, the slower it will be due to inter-GPU communication overhead. If you can fit the model on a single GPU, it is recommended to do so.
- `TENSOR_PARALLEL_SIZE`: Number of GPUs to shard the model across (default: `1`). - `TENSOR_PARALLEL_SIZE`: Number of GPUs to shard the model across (default: `1`).
- If you are having issues loading your model with Tensor Parallelism, try decreasing `VLLM_CPU_FRACTION` (default: `1`). - If you are having issues loading your model with Tensor Parallelism, try decreasing `VLLM_CPU_FRACTION` (default: `1`).
- System Settings: - System Settings:
- `GPU_MEMORY_UTILIZATION`: GPU VRAM utilization (default: `0.98`). - `GPU_MEMORY_UTILIZATION`: GPU VRAM utilization (default: `0.98`).
- `MAX_PARALLEL_LOADING_WORKERS`: Maximum number of parallel workers for loading models (default: `number of available CPU cores` if `TENSOR_PARALLEL_SIZE` is `1`, otherwise `None`). - `MAX_PARALLEL_LOADING_WORKERS`: Maximum number of parallel workers for loading models, for non-Tensor Parallel only. (default: `number of available CPU cores` if `TENSOR_PARALLEL_SIZE` is `1`, otherwise `None`).
- Serverless Settings: - Serverless Settings:
@@ -96,9 +99,12 @@ To build an image with the model baked in, you must specify the following docker
- **Required** - **Required**
- `MODEL_NAME` - `MODEL_NAME`
- **Optional** - **Optional**
- `MODEL_REVISION`: Model revision to load (default: `main`).
- `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't. If your intention is to bake the model into the image, you should set this to something like `/models` to make sure there are no issues if you were to accidentally attach network storage.) - `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't. If your intention is to bake the model into the image, you should set this to something like `/models` to make sure there are no issues if you were to accidentally attach network storage.)
- `QUANTIZATION` - `QUANTIZATION`
- `WORKER_CUDA_VERSION`: `11.8.0` or `12.1.0` (default: `11.8.0` due to a small amount of workers not having CUDA 12.1 support yet. `12.1.0` is recommended for optimal performance). - `WORKER_CUDA_VERSION`: `11.8.0` or `12.1.0` (default: `11.8.0` due to a small amount of workers not having CUDA 12.1 support yet. `12.1.0` is recommended for optimal performance).
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer)
- `TOKENIZER_REVISION`: Tokenizer revision to load (default: `main`).
For the remaining settings, you may apply them as environment variables when running the container. Supported environment variables are listed in the [Environment Variables](#environment-variables) section. For the remaining settings, you may apply them as environment variables when running the container. Supported environment variables are listed in the [Environment Variables](#environment-variables) section.
+50
View File
@@ -0,0 +1,50 @@
import os
import shutil
from huggingface_hub import snapshot_download
from vllm.model_executor.weight_utils import prepare_hf_model_weights, Disabledtqdm
def download_extras_or_tokenizer(model_name, cache_dir, revision, extras=False):
"""Download model or tokenizer and prepare its weights, returning the local folder path."""
pattern = ["*token*", "*.json"] if extras else None
extra_dir = "/extras" if extras else ""
folder = snapshot_download(
model_name,
cache_dir=cache_dir + extra_dir,
revision=revision,
tqdm_class=Disabledtqdm,
allow_patterns=pattern if extras else None,
ignore_patterns=["*.safetensors", "*.bin", "*.pt"] if not extras else None
)
return folder
def move_files(src_dir, dest_dir):
"""Move files from source to destination directory."""
for f in os.listdir(src_dir):
src_path = os.path.join(src_dir, f)
dst_path = os.path.join(dest_dir, f)
shutil.copy2(src_path, dst_path)
os.remove(src_path)
if __name__ == "__main__":
model, download_dir = os.getenv("MODEL_NAME"), os.getenv("HF_HOME")
tokenizer = os.getenv("TOKENIZER_NAME", model)
revisions = {
"model": os.getenv("MODEL_REVISION") or None,
"tokenizer": os.getenv("TOKENIZER_REVISION") or None
}
if not model or not download_dir:
raise ValueError(f"Must specify model and download_dir. Model: {model}, download_dir: {download_dir}")
os.makedirs(download_dir, exist_ok=True)
model_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights(model_name_or_path=model, revision=revisions["model"], cache_dir=download_dir)
model_extras_folder = download_extras_or_tokenizer(model, download_dir, revisions["model"], extras=True)
move_files(model_extras_folder, model_folder)
with open("/local_model_path.txt", "w") as f:
f.write(model_folder)
if tokenizer != model:
tokenizer_folder = download_extras_or_tokenizer(tokenizer, download_dir, revisions["tokenizer"])
with open("/local_tokenizer_path.txt", "w") as f:
f.write(tokenizer_folder)
-25
View File
@@ -1,25 +0,0 @@
import os
import logging
from vllm.model_executor.weight_utils import prepare_hf_model_weights
if __name__ == "__main__":
model = os.getenv("MODEL_NAME")
download_dir = os.getenv("HF_HOME")
if not model or not download_dir:
raise ValueError(f"Must specify model and download_dir. Model: {model}, download_dir: {download_dir}")
if not os.path.exists(download_dir):
os.makedirs(download_dir)
logging.info(f"Downloading model {model} to {download_dir}")
hf_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights(
model_name_or_path=model,
cache_dir=download_dir,
)
logging.info(f"Finished downloading model {model} to {download_dir}")
# Wrie hf_folder to file
with open("/local_model_path.txt", "w") as f:
f.write(hf_folder)
+22 -10
View File
@@ -13,8 +13,8 @@ from dotenv import load_dotenv
class Tokenizer: class Tokenizer:
def __init__(self, model_name): def __init__(self, tokenizer_name_or_path, tokenizer_revision):
self.tokenizer = AutoTokenizer.from_pretrained(model_name) self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_name_or_path, revision=tokenizer_revision)
self.custom_chat_template = os.getenv("CUSTOM_CHAT_TEMPLATE") self.custom_chat_template = os.getenv("CUSTOM_CHAT_TEMPLATE")
self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template) self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template)
if self.custom_chat_template and isinstance(self.custom_chat_template, str): if self.custom_chat_template and isinstance(self.custom_chat_template, str):
@@ -41,7 +41,7 @@ class vLLMEngine:
load_dotenv() # For local development load_dotenv() # For local development
self.config = self._initialize_config() self.config = self._initialize_config()
logging.info("vLLM config: %s", self.config) logging.info("vLLM config: %s", self.config)
self.tokenizer = Tokenizer(os.getenv("TOKENIZER_NAME", os.getenv("MODEL_NAME"))) self.tokenizer = Tokenizer(self.config["tokenizer"], self.config["tokenizer_revision"])
self.llm = self._initialize_llm() if engine is None else engine self.llm = self._initialize_llm() if engine is None else engine
self.openai_engine = self._initialize_openai() self.openai_engine = self._initialize_openai()
self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY)) self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
@@ -60,7 +60,6 @@ class vLLMEngine:
yield batch yield batch
async def generate_vllm(self, llm_input, validated_sampling_params, batch_size, stream, apply_chat_template, request_id: str) -> AsyncGenerator[dict, None]: async def generate_vllm(self, llm_input, validated_sampling_params, batch_size, stream, apply_chat_template, request_id: str) -> AsyncGenerator[dict, None]:
if apply_chat_template or isinstance(llm_input, list): if apply_chat_template or isinstance(llm_input, list):
llm_input = self.tokenizer.apply_chat_template(llm_input) llm_input = self.tokenizer.apply_chat_template(llm_input)
validated_sampling_params = SamplingParams(**validated_sampling_params) validated_sampling_params = SamplingParams(**validated_sampling_params)
@@ -165,15 +164,20 @@ class vLLMEngine:
def _initialize_config(self): def _initialize_config(self):
quantization = self._get_quantization() quantization = self._get_quantization()
model, download_dir = self._get_model_name_and_path() model, download_dir, model_revision = self._get_model_info()
tokenizer_name_or_path, tokenizer_revision = self._get_tokenizer_info()
if not tokenizer_name_or_path:
tokenizer_name_or_path = model
return { return {
"model": model, "model": model,
"revision": model_revision,
"download_dir": download_dir, "download_dir": download_dir,
"quantization": quantization, "quantization": quantization,
"load_format": os.getenv("LOAD_FORMAT", "auto"), "load_format": os.getenv("LOAD_FORMAT", "auto"),
"dtype": "half" if quantization else "auto", "dtype": "half" if quantization else "auto",
"tokenizer": os.getenv("TOKENIZER_NAME"), "tokenizer": tokenizer_name_or_path,
"tokenizer_revision": tokenizer_revision,
"disable_log_stats": bool(int(os.getenv("DISABLE_LOG_STATS", 1))), "disable_log_stats": bool(int(os.getenv("DISABLE_LOG_STATS", 1))),
"disable_log_requests": bool(int(os.getenv("DISABLE_LOG_REQUESTS", 1))), "disable_log_requests": bool(int(os.getenv("DISABLE_LOG_REQUESTS", 1))),
"trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))), "trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))),
@@ -202,13 +206,21 @@ class vLLMEngine:
else: else:
return int(os.getenv("MAX_PARALLEL_LOADING_WORKERS", count_physical_cores())) return int(os.getenv("MAX_PARALLEL_LOADING_WORKERS", count_physical_cores()))
def _get_model_name_and_path(self): def _get_model_info(self):
if os.path.exists("/local_model_path.txt"): if os.path.exists("/local_model_path.txt"):
model, download_dir = open("/local_model_path.txt", "r").read().strip(), None model, download_dir, revision = open("/local_model_path.txt", "r").read().strip(), None, None
logging.info("Using local model at %s", model) logging.info("Using local model at %s", model)
else: else:
model, download_dir = os.getenv("MODEL_NAME"), os.getenv("HF_HOME") model, download_dir, revision = os.getenv("MODEL_NAME"), os.getenv("HF_HOME"), os.getenv("MODEL_REVISION") or None
return model, download_dir return model, download_dir, revision
def _get_tokenizer_info(self):
if os.path.exists("/local_tokenizer_path.txt"):
tokenizer_name_or_path, revision = open("/local_tokenizer_path.txt", "r").read().strip(), None
logging.info("Using local tokenizer at %s", tokenizer_name_or_path)
else:
tokenizer_name_or_path, revision = os.getenv("TOKENIZER_NAME"), os.getenv("TOKENIZER_REVISION") or None
return tokenizer_name_or_path, revision
def _get_num_gpu_shard(self): def _get_num_gpu_shard(self):
num_gpu_shard = int(os.getenv("TENSOR_PARALLEL_SIZE", 1)) num_gpu_shard = int(os.getenv("TENSOR_PARALLEL_SIZE", 1))