Added download of model extras into weights folder, separate download of tokenizer, making engine.py utilize downloaded tokenizer, model and tokenizer revision
This commit is contained in:
@@ -15,10 +15,16 @@ COPY src /src
|
|||||||
|
|
||||||
# Setup for Option 2: Building the Image with the Model included
|
# Setup for Option 2: Building the Image with the Model included
|
||||||
ARG MODEL_NAME=""
|
ARG MODEL_NAME=""
|
||||||
|
ARG TOKENIZER_NAME=""
|
||||||
ARG BASE_PATH="/runpod-volume"
|
ARG BASE_PATH="/runpod-volume"
|
||||||
ARG QUANTIZATION=""
|
ARG QUANTIZATION=""
|
||||||
|
ARG MODEL_REVISION="main"
|
||||||
|
ARG TOKENIZER_REVISION="main"
|
||||||
|
|
||||||
ENV MODEL_NAME=$MODEL_NAME \
|
ENV MODEL_NAME=$MODEL_NAME \
|
||||||
|
MODEL_REVISION=$REVISION \
|
||||||
|
TOKENIZER_NAME=$TOKENIZER_NAME \
|
||||||
|
TOKENIZER_REVISION=$TOKENIZER_REVISION \
|
||||||
BASE_PATH=$BASE_PATH \
|
BASE_PATH=$BASE_PATH \
|
||||||
QUANTIZATION=$QUANTIZATION \
|
QUANTIZATION=$QUANTIZATION \
|
||||||
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
|
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
|
||||||
|
|||||||
@@ -58,8 +58,7 @@ Development Image: ```runpod/worker-vllm:dev```
|
|||||||
|
|
||||||
**Optional**:
|
**Optional**:
|
||||||
- LLM Settings:
|
- LLM Settings:
|
||||||
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`)
|
- `MODEL_REVISION`: Model revision to load (default: `main`).
|
||||||
- `CUSTOM_CHAT_TEMPLATE`: Custom chat jinja template, read more about Hugging Face chat templates [here](https://huggingface.co/docs/transformers/chat_templating). (default: `None`)
|
|
||||||
- `MAX_MODEL_LENGTH`: Maximum number of tokens for the engine to be able to handle. (default: maximum supported by the model)
|
- `MAX_MODEL_LENGTH`: Maximum number of tokens for the engine to be able to handle. (default: maximum supported by the model)
|
||||||
- `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't)
|
- `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't)
|
||||||
- `LOAD_FORMAT`: Format to load model in (default: `auto`).
|
- `LOAD_FORMAT`: Format to load model in (default: `auto`).
|
||||||
@@ -67,8 +66,12 @@ Development Image: ```runpod/worker-vllm:dev```
|
|||||||
- `QUANTIZATION`: AWQ (`awq`), SqueezeLLM (`squeezellm`) or GPTQ (`gptq`) Quantization. The specified Model Repo must be of a quantized model. (default: `None`)
|
- `QUANTIZATION`: AWQ (`awq`), SqueezeLLM (`squeezellm`) or GPTQ (`gptq`) Quantization. The specified Model Repo must be of a quantized model. (default: `None`)
|
||||||
- `TRUST_REMOTE_CODE`: Trust remote code for Hugging Face (default: `0`)
|
- `TRUST_REMOTE_CODE`: Trust remote code for Hugging Face (default: `0`)
|
||||||
|
|
||||||
- Tensor Parallelism:
|
- Tokenizer Settings:
|
||||||
|
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer)
|
||||||
|
- `TOKENIZER_REVISION`: Tokenizer revision to load (default: `main`).
|
||||||
|
- `CUSTOM_CHAT_TEMPLATE`: Custom chat jinja template, read more about Hugging Face chat templates [here](https://huggingface.co/docs/transformers/chat_templating). (default: `None`)
|
||||||
|
|
||||||
|
- Tensor Parallelism:
|
||||||
Note that the more GPUs you split a model's weights accross, the slower it will be due to inter-GPU communication overhead. If you can fit the model on a single GPU, it is recommended to do so.
|
Note that the more GPUs you split a model's weights accross, the slower it will be due to inter-GPU communication overhead. If you can fit the model on a single GPU, it is recommended to do so.
|
||||||
- `TENSOR_PARALLEL_SIZE`: Number of GPUs to shard the model across (default: `1`).
|
- `TENSOR_PARALLEL_SIZE`: Number of GPUs to shard the model across (default: `1`).
|
||||||
- If you are having issues loading your model with Tensor Parallelism, try decreasing `VLLM_CPU_FRACTION` (default: `1`).
|
- If you are having issues loading your model with Tensor Parallelism, try decreasing `VLLM_CPU_FRACTION` (default: `1`).
|
||||||
@@ -96,9 +99,12 @@ To build an image with the model baked in, you must specify the following docker
|
|||||||
- **Required**
|
- **Required**
|
||||||
- `MODEL_NAME`
|
- `MODEL_NAME`
|
||||||
- **Optional**
|
- **Optional**
|
||||||
|
- `MODEL_REVISION`: Model revision to load (default: `main`).
|
||||||
- `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't. If your intention is to bake the model into the image, you should set this to something like `/models` to make sure there are no issues if you were to accidentally attach network storage.)
|
- `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't. If your intention is to bake the model into the image, you should set this to something like `/models` to make sure there are no issues if you were to accidentally attach network storage.)
|
||||||
- `QUANTIZATION`
|
- `QUANTIZATION`
|
||||||
- `WORKER_CUDA_VERSION`: `11.8.0` or `12.1.0` (default: `11.8.0` due to a small amount of workers not having CUDA 12.1 support yet. `12.1.0` is recommended for optimal performance).
|
- `WORKER_CUDA_VERSION`: `11.8.0` or `12.1.0` (default: `11.8.0` due to a small amount of workers not having CUDA 12.1 support yet. `12.1.0` is recommended for optimal performance).
|
||||||
|
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer)
|
||||||
|
- `TOKENIZER_REVISION`: Tokenizer revision to load (default: `main`).
|
||||||
|
|
||||||
For the remaining settings, you may apply them as environment variables when running the container. Supported environment variables are listed in the [Environment Variables](#environment-variables) section.
|
For the remaining settings, you may apply them as environment variables when running the container. Supported environment variables are listed in the [Environment Variables](#environment-variables) section.
|
||||||
|
|
||||||
|
|||||||
+37
-13
@@ -1,11 +1,16 @@
|
|||||||
import os
|
import os
|
||||||
|
import shutil
|
||||||
import logging
|
import logging
|
||||||
from huggingface_hub import snapshot_download
|
from huggingface_hub import snapshot_download
|
||||||
from vllm.model_executor.weight_utils import prepare_hf_model_weights
|
from vllm.model_executor.weight_utils import Disabledtqdm, prepare_hf_model_weights
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
model = os.getenv("MODEL_NAME")
|
model = os.getenv("MODEL_NAME")
|
||||||
download_dir = os.getenv("HF_HOME")
|
download_dir = os.getenv("HF_HOME")
|
||||||
|
tokenizer = os.getenv("TOKENIZER_NAME", model)
|
||||||
|
model_revision = os.getenv("MODEL_REVISION", "main")
|
||||||
|
tokenizer_revision = os.getenv("TOKENIZER_REVISION", "main")
|
||||||
|
|
||||||
if not model or not download_dir:
|
if not model or not download_dir:
|
||||||
raise ValueError(f"Must specify model and download_dir. Model: {model}, download_dir: {download_dir}")
|
raise ValueError(f"Must specify model and download_dir. Model: {model}, download_dir: {download_dir}")
|
||||||
|
|
||||||
@@ -14,22 +19,41 @@ if __name__ == "__main__":
|
|||||||
|
|
||||||
logging.info(f"Downloading model {model} to {download_dir}")
|
logging.info(f"Downloading model {model} to {download_dir}")
|
||||||
|
|
||||||
hf_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights(
|
model_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights(
|
||||||
model_name_or_path=model,
|
model_name_or_path=model,
|
||||||
cache_dir=download_dir,
|
cache_dir=download_dir,
|
||||||
)
|
revision=model_revision,
|
||||||
|
|
||||||
snapshot_download(
|
|
||||||
model,
|
|
||||||
cache_dir=download_dir,
|
|
||||||
allow_patterns=[
|
|
||||||
"*token*",
|
|
||||||
"config.json",
|
|
||||||
]
|
|
||||||
)
|
)
|
||||||
|
|
||||||
logging.info(f"Finished downloading model {model} to {download_dir}")
|
model_extras_folder = snapshot_download(model,
|
||||||
|
allow_patterns=[
|
||||||
|
"*token*",
|
||||||
|
"*.json"
|
||||||
|
],
|
||||||
|
cache_dir=download_dir + "/extras",
|
||||||
|
tqdm_class=Disabledtqdm,
|
||||||
|
revision=model_revision)
|
||||||
|
|
||||||
|
# Move extras to hf_folder
|
||||||
|
for f in os.listdir(model_extras_folder):
|
||||||
|
shutil.move(model_extras_folder + "/" + f, model_folder + "/" + f)
|
||||||
|
|
||||||
# Wrie hf_folder to file
|
# Wrie hf_folder to file
|
||||||
with open("/local_model_path.txt", "w") as f:
|
with open("/local_model_path.txt", "w") as f:
|
||||||
f.write(hf_folder)
|
f.write(model_folder)
|
||||||
|
|
||||||
|
logging.info(f"Finished downloading model {model} to {download_dir}")
|
||||||
|
|
||||||
|
if tokenizer != model:
|
||||||
|
logging.info(f"Downloading tokenizer {tokenizer} to {download_dir}")
|
||||||
|
tokenizer_folder = snapshot_download(tokenizer,
|
||||||
|
cache_dir=download_dir,
|
||||||
|
tqdm_class=Disabledtqdm,
|
||||||
|
revision=tokenizer_revision)
|
||||||
|
with open("/local_tokenizer_path.txt", "w") as f:
|
||||||
|
f.write(tokenizer_folder)
|
||||||
|
|
||||||
|
logging.info(f"Finished downloading tokenizer {tokenizer} to {download_dir}")
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+13
-2
@@ -41,7 +41,7 @@ class vLLMEngine:
|
|||||||
load_dotenv() # For local development
|
load_dotenv() # For local development
|
||||||
self.config = self._initialize_config()
|
self.config = self._initialize_config()
|
||||||
logging.info("vLLM config: %s", self.config)
|
logging.info("vLLM config: %s", self.config)
|
||||||
self.tokenizer = Tokenizer(os.getenv("TOKENIZER_NAME", os.getenv("MODEL_NAME")))
|
self.tokenizer = Tokenizer(self.config["tokenizer_name_or_path"])
|
||||||
self.llm = self._initialize_llm() if engine is None else engine
|
self.llm = self._initialize_llm() if engine is None else engine
|
||||||
self.openai_engine = self._initialize_openai()
|
self.openai_engine = self._initialize_openai()
|
||||||
self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
|
self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
|
||||||
@@ -166,6 +166,9 @@ class vLLMEngine:
|
|||||||
def _initialize_config(self):
|
def _initialize_config(self):
|
||||||
quantization = self._get_quantization()
|
quantization = self._get_quantization()
|
||||||
model, download_dir = self._get_model_name_and_path()
|
model, download_dir = self._get_model_name_and_path()
|
||||||
|
tokenizer_name_or_path = self._get_tokenizer_name_or_path()
|
||||||
|
if not tokenizer_name_or_path:
|
||||||
|
tokenizer_name_or_path = model
|
||||||
|
|
||||||
return {
|
return {
|
||||||
"model": model,
|
"model": model,
|
||||||
@@ -173,7 +176,7 @@ class vLLMEngine:
|
|||||||
"quantization": quantization,
|
"quantization": quantization,
|
||||||
"load_format": os.getenv("LOAD_FORMAT", "auto"),
|
"load_format": os.getenv("LOAD_FORMAT", "auto"),
|
||||||
"dtype": "half" if quantization else "auto",
|
"dtype": "half" if quantization else "auto",
|
||||||
"tokenizer": os.getenv("TOKENIZER_NAME"),
|
"tokenizer": tokenizer_name_or_path,
|
||||||
"disable_log_stats": bool(int(os.getenv("DISABLE_LOG_STATS", 1))),
|
"disable_log_stats": bool(int(os.getenv("DISABLE_LOG_STATS", 1))),
|
||||||
"disable_log_requests": bool(int(os.getenv("DISABLE_LOG_REQUESTS", 1))),
|
"disable_log_requests": bool(int(os.getenv("DISABLE_LOG_REQUESTS", 1))),
|
||||||
"trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))),
|
"trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))),
|
||||||
@@ -209,6 +212,14 @@ class vLLMEngine:
|
|||||||
else:
|
else:
|
||||||
model, download_dir = os.getenv("MODEL_NAME"), os.getenv("HF_HOME")
|
model, download_dir = os.getenv("MODEL_NAME"), os.getenv("HF_HOME")
|
||||||
return model, download_dir
|
return model, download_dir
|
||||||
|
|
||||||
|
def _get_tokenizer_name_or_path(self):
|
||||||
|
if os.path.exists("/local_tokenizer_path.txt"):
|
||||||
|
tokenizer_name_or_path = open("/local_tokenizer_path.txt", "r").read().strip()
|
||||||
|
logging.info("Using local tokenizer at %s", tokenizer_name_or_path)
|
||||||
|
else:
|
||||||
|
tokenizer_name_or_path = os.getenv("TOKENIZER_NAME")
|
||||||
|
return tokenizer_name_or_path
|
||||||
|
|
||||||
def _get_num_gpu_shard(self):
|
def _get_num_gpu_shard(self):
|
||||||
num_gpu_shard = int(os.getenv("TENSOR_PARALLEL_SIZE", 1))
|
num_gpu_shard = int(os.getenv("TENSOR_PARALLEL_SIZE", 1))
|
||||||
|
|||||||
Reference in New Issue
Block a user