Added download of model extras into weights folder, separate download of tokenizer, making engine.py utilize downloaded tokenizer, model and tokenizer revision

This commit is contained in:
alpayariyak
2024-01-31 22:57:32 -05:00
parent f4d7c75504
commit b1720a154d
4 changed files with 65 additions and 18 deletions
+6
View File
@@ -15,10 +15,16 @@ COPY src /src
# Setup for Option 2: Building the Image with the Model included # Setup for Option 2: Building the Image with the Model included
ARG MODEL_NAME="" ARG MODEL_NAME=""
ARG TOKENIZER_NAME=""
ARG BASE_PATH="/runpod-volume" ARG BASE_PATH="/runpod-volume"
ARG QUANTIZATION="" ARG QUANTIZATION=""
ARG MODEL_REVISION="main"
ARG TOKENIZER_REVISION="main"
ENV MODEL_NAME=$MODEL_NAME \ ENV MODEL_NAME=$MODEL_NAME \
MODEL_REVISION=$REVISION \
TOKENIZER_NAME=$TOKENIZER_NAME \
TOKENIZER_REVISION=$TOKENIZER_REVISION \
BASE_PATH=$BASE_PATH \ BASE_PATH=$BASE_PATH \
QUANTIZATION=$QUANTIZATION \ QUANTIZATION=$QUANTIZATION \
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \ HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
+9 -3
View File
@@ -58,8 +58,7 @@ Development Image: ```runpod/worker-vllm:dev```
**Optional**: **Optional**:
- LLM Settings: - LLM Settings:
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`) - `MODEL_REVISION`: Model revision to load (default: `main`).
- `CUSTOM_CHAT_TEMPLATE`: Custom chat jinja template, read more about Hugging Face chat templates [here](https://huggingface.co/docs/transformers/chat_templating). (default: `None`)
- `MAX_MODEL_LENGTH`: Maximum number of tokens for the engine to be able to handle. (default: maximum supported by the model) - `MAX_MODEL_LENGTH`: Maximum number of tokens for the engine to be able to handle. (default: maximum supported by the model)
- `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't) - `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't)
- `LOAD_FORMAT`: Format to load model in (default: `auto`). - `LOAD_FORMAT`: Format to load model in (default: `auto`).
@@ -67,8 +66,12 @@ Development Image: ```runpod/worker-vllm:dev```
- `QUANTIZATION`: AWQ (`awq`), SqueezeLLM (`squeezellm`) or GPTQ (`gptq`) Quantization. The specified Model Repo must be of a quantized model. (default: `None`) - `QUANTIZATION`: AWQ (`awq`), SqueezeLLM (`squeezellm`) or GPTQ (`gptq`) Quantization. The specified Model Repo must be of a quantized model. (default: `None`)
- `TRUST_REMOTE_CODE`: Trust remote code for Hugging Face (default: `0`) - `TRUST_REMOTE_CODE`: Trust remote code for Hugging Face (default: `0`)
- Tensor Parallelism: - Tokenizer Settings:
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer)
- `TOKENIZER_REVISION`: Tokenizer revision to load (default: `main`).
- `CUSTOM_CHAT_TEMPLATE`: Custom chat jinja template, read more about Hugging Face chat templates [here](https://huggingface.co/docs/transformers/chat_templating). (default: `None`)
- Tensor Parallelism:
Note that the more GPUs you split a model's weights accross, the slower it will be due to inter-GPU communication overhead. If you can fit the model on a single GPU, it is recommended to do so. Note that the more GPUs you split a model's weights accross, the slower it will be due to inter-GPU communication overhead. If you can fit the model on a single GPU, it is recommended to do so.
- `TENSOR_PARALLEL_SIZE`: Number of GPUs to shard the model across (default: `1`). - `TENSOR_PARALLEL_SIZE`: Number of GPUs to shard the model across (default: `1`).
- If you are having issues loading your model with Tensor Parallelism, try decreasing `VLLM_CPU_FRACTION` (default: `1`). - If you are having issues loading your model with Tensor Parallelism, try decreasing `VLLM_CPU_FRACTION` (default: `1`).
@@ -96,9 +99,12 @@ To build an image with the model baked in, you must specify the following docker
- **Required** - **Required**
- `MODEL_NAME` - `MODEL_NAME`
- **Optional** - **Optional**
- `MODEL_REVISION`: Model revision to load (default: `main`).
- `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't. If your intention is to bake the model into the image, you should set this to something like `/models` to make sure there are no issues if you were to accidentally attach network storage.) - `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't. If your intention is to bake the model into the image, you should set this to something like `/models` to make sure there are no issues if you were to accidentally attach network storage.)
- `QUANTIZATION` - `QUANTIZATION`
- `WORKER_CUDA_VERSION`: `11.8.0` or `12.1.0` (default: `11.8.0` due to a small amount of workers not having CUDA 12.1 support yet. `12.1.0` is recommended for optimal performance). - `WORKER_CUDA_VERSION`: `11.8.0` or `12.1.0` (default: `11.8.0` due to a small amount of workers not having CUDA 12.1 support yet. `12.1.0` is recommended for optimal performance).
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer)
- `TOKENIZER_REVISION`: Tokenizer revision to load (default: `main`).
For the remaining settings, you may apply them as environment variables when running the container. Supported environment variables are listed in the [Environment Variables](#environment-variables) section. For the remaining settings, you may apply them as environment variables when running the container. Supported environment variables are listed in the [Environment Variables](#environment-variables) section.
+37 -13
View File
@@ -1,11 +1,16 @@
import os import os
import shutil
import logging import logging
from huggingface_hub import snapshot_download from huggingface_hub import snapshot_download
from vllm.model_executor.weight_utils import prepare_hf_model_weights from vllm.model_executor.weight_utils import Disabledtqdm, prepare_hf_model_weights
if __name__ == "__main__": if __name__ == "__main__":
model = os.getenv("MODEL_NAME") model = os.getenv("MODEL_NAME")
download_dir = os.getenv("HF_HOME") download_dir = os.getenv("HF_HOME")
tokenizer = os.getenv("TOKENIZER_NAME", model)
model_revision = os.getenv("MODEL_REVISION", "main")
tokenizer_revision = os.getenv("TOKENIZER_REVISION", "main")
if not model or not download_dir: if not model or not download_dir:
raise ValueError(f"Must specify model and download_dir. Model: {model}, download_dir: {download_dir}") raise ValueError(f"Must specify model and download_dir. Model: {model}, download_dir: {download_dir}")
@@ -14,22 +19,41 @@ if __name__ == "__main__":
logging.info(f"Downloading model {model} to {download_dir}") logging.info(f"Downloading model {model} to {download_dir}")
hf_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights( model_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights(
model_name_or_path=model, model_name_or_path=model,
cache_dir=download_dir, cache_dir=download_dir,
) revision=model_revision,
snapshot_download(
model,
cache_dir=download_dir,
allow_patterns=[
"*token*",
"config.json",
]
) )
logging.info(f"Finished downloading model {model} to {download_dir}") model_extras_folder = snapshot_download(model,
allow_patterns=[
"*token*",
"*.json"
],
cache_dir=download_dir + "/extras",
tqdm_class=Disabledtqdm,
revision=model_revision)
# Move extras to hf_folder
for f in os.listdir(model_extras_folder):
shutil.move(model_extras_folder + "/" + f, model_folder + "/" + f)
# Wrie hf_folder to file # Wrie hf_folder to file
with open("/local_model_path.txt", "w") as f: with open("/local_model_path.txt", "w") as f:
f.write(hf_folder) f.write(model_folder)
logging.info(f"Finished downloading model {model} to {download_dir}")
if tokenizer != model:
logging.info(f"Downloading tokenizer {tokenizer} to {download_dir}")
tokenizer_folder = snapshot_download(tokenizer,
cache_dir=download_dir,
tqdm_class=Disabledtqdm,
revision=tokenizer_revision)
with open("/local_tokenizer_path.txt", "w") as f:
f.write(tokenizer_folder)
logging.info(f"Finished downloading tokenizer {tokenizer} to {download_dir}")
+13 -2
View File
@@ -41,7 +41,7 @@ class vLLMEngine:
load_dotenv() # For local development load_dotenv() # For local development
self.config = self._initialize_config() self.config = self._initialize_config()
logging.info("vLLM config: %s", self.config) logging.info("vLLM config: %s", self.config)
self.tokenizer = Tokenizer(os.getenv("TOKENIZER_NAME", os.getenv("MODEL_NAME"))) self.tokenizer = Tokenizer(self.config["tokenizer_name_or_path"])
self.llm = self._initialize_llm() if engine is None else engine self.llm = self._initialize_llm() if engine is None else engine
self.openai_engine = self._initialize_openai() self.openai_engine = self._initialize_openai()
self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY)) self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
@@ -166,6 +166,9 @@ class vLLMEngine:
def _initialize_config(self): def _initialize_config(self):
quantization = self._get_quantization() quantization = self._get_quantization()
model, download_dir = self._get_model_name_and_path() model, download_dir = self._get_model_name_and_path()
tokenizer_name_or_path = self._get_tokenizer_name_or_path()
if not tokenizer_name_or_path:
tokenizer_name_or_path = model
return { return {
"model": model, "model": model,
@@ -173,7 +176,7 @@ class vLLMEngine:
"quantization": quantization, "quantization": quantization,
"load_format": os.getenv("LOAD_FORMAT", "auto"), "load_format": os.getenv("LOAD_FORMAT", "auto"),
"dtype": "half" if quantization else "auto", "dtype": "half" if quantization else "auto",
"tokenizer": os.getenv("TOKENIZER_NAME"), "tokenizer": tokenizer_name_or_path,
"disable_log_stats": bool(int(os.getenv("DISABLE_LOG_STATS", 1))), "disable_log_stats": bool(int(os.getenv("DISABLE_LOG_STATS", 1))),
"disable_log_requests": bool(int(os.getenv("DISABLE_LOG_REQUESTS", 1))), "disable_log_requests": bool(int(os.getenv("DISABLE_LOG_REQUESTS", 1))),
"trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))), "trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))),
@@ -209,6 +212,14 @@ class vLLMEngine:
else: else:
model, download_dir = os.getenv("MODEL_NAME"), os.getenv("HF_HOME") model, download_dir = os.getenv("MODEL_NAME"), os.getenv("HF_HOME")
return model, download_dir return model, download_dir
def _get_tokenizer_name_or_path(self):
if os.path.exists("/local_tokenizer_path.txt"):
tokenizer_name_or_path = open("/local_tokenizer_path.txt", "r").read().strip()
logging.info("Using local tokenizer at %s", tokenizer_name_or_path)
else:
tokenizer_name_or_path = os.getenv("TOKENIZER_NAME")
return tokenizer_name_or_path
def _get_num_gpu_shard(self): def _get_num_gpu_shard(self):
num_gpu_shard = int(os.getenv("TENSOR_PARALLEL_SIZE", 1)) num_gpu_shard = int(os.getenv("TENSOR_PARALLEL_SIZE", 1))