From 664dd35782d75ed82b42e0c347c0202ca9f58851 Mon Sep 17 00:00:00 2001 From: Casper Date: Wed, 31 Jan 2024 17:59:52 +0100 Subject: [PATCH 1/6] Download tokenizer upon build --- src/download_model.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/src/download_model.py b/src/download_model.py index 086a2a0..f0f2d79 100644 --- a/src/download_model.py +++ b/src/download_model.py @@ -1,5 +1,6 @@ import os import logging +from transformers import AutoTokenizer from vllm.model_executor.weight_utils import prepare_hf_model_weights if __name__ == "__main__": @@ -17,6 +18,12 @@ if __name__ == "__main__": model_name_or_path=model, cache_dir=download_dir, ) + + tokenizer = os.getenv("TOKENIZER_NAME", model) + AutoTokenizer.from_pretrained( + pretrained_model_name_or_path=tokenizer, + cache_dir=download_dir, + ) logging.info(f"Finished downloading model {model} to {download_dir}") From fd00a1ece34e67c26d756515de8a91b2c786c678 Mon Sep 17 00:00:00 2001 From: Casper Date: Wed, 31 Jan 2024 18:24:57 +0100 Subject: [PATCH 2/6] Update to use snapshot_download --- src/download_model.py | 16 ++++++---------- 1 file changed, 6 insertions(+), 10 deletions(-) diff --git a/src/download_model.py b/src/download_model.py index f0f2d79..ce8c9a4 100644 --- a/src/download_model.py +++ b/src/download_model.py @@ -1,7 +1,7 @@ import os import logging from transformers import AutoTokenizer -from vllm.model_executor.weight_utils import prepare_hf_model_weights +from huggingface_hub import snapshot_download if __name__ == "__main__": model = os.getenv("MODEL_NAME") @@ -13,16 +13,12 @@ if __name__ == "__main__": os.makedirs(download_dir) logging.info(f"Downloading model {model} to {download_dir}") - - hf_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights( - model_name_or_path=model, - cache_dir=download_dir, - ) - - tokenizer = os.getenv("TOKENIZER_NAME", model) - AutoTokenizer.from_pretrained( - pretrained_model_name_or_path=tokenizer, + + hf_folder = snapshot_download( + model, + local_dir=download_dir, cache_dir=download_dir, + local_dir_use_symlinks=False, ) logging.info(f"Finished downloading model {model} to {download_dir}") From 3adc9e333653dcccac741b60066d857ea38209b6 Mon Sep 17 00:00:00 2001 From: Casper Date: Wed, 31 Jan 2024 18:27:49 +0100 Subject: [PATCH 3/6] Remove unused import --- src/download_model.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/src/download_model.py b/src/download_model.py index ce8c9a4..8dc73f4 100644 --- a/src/download_model.py +++ b/src/download_model.py @@ -1,6 +1,5 @@ import os import logging -from transformers import AutoTokenizer from huggingface_hub import snapshot_download if __name__ == "__main__": @@ -13,7 +12,7 @@ if __name__ == "__main__": os.makedirs(download_dir) logging.info(f"Downloading model {model} to {download_dir}") - + hf_folder = snapshot_download( model, local_dir=download_dir, From f4d7c755047534dfb2b4a0fa424899afdac73bfe Mon Sep 17 00:00:00 2001 From: Casper Date: Wed, 31 Jan 2024 22:16:44 +0100 Subject: [PATCH 4/6] Snapshot download only tokenizer/config related things --- src/download_model.py | 18 +++++++++++++----- 1 file changed, 13 insertions(+), 5 deletions(-) diff --git a/src/download_model.py b/src/download_model.py index 8dc73f4..12850f9 100644 --- a/src/download_model.py +++ b/src/download_model.py @@ -1,6 +1,7 @@ import os import logging from huggingface_hub import snapshot_download +from vllm.model_executor.weight_utils import prepare_hf_model_weights if __name__ == "__main__": model = os.getenv("MODEL_NAME") @@ -12,12 +13,19 @@ if __name__ == "__main__": os.makedirs(download_dir) logging.info(f"Downloading model {model} to {download_dir}") - - hf_folder = snapshot_download( - model, - local_dir=download_dir, + + hf_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights( + model_name_or_path=model, cache_dir=download_dir, - local_dir_use_symlinks=False, + ) + + snapshot_download( + model, + cache_dir=download_dir, + allow_patterns=[ + "*token*", + "config.json", + ] ) logging.info(f"Finished downloading model {model} to {download_dir}") From b1720a154d0de02afb319a720222b9f3f3091852 Mon Sep 17 00:00:00 2001 From: alpayariyak Date: Wed, 31 Jan 2024 22:57:32 -0500 Subject: [PATCH 5/6] Added download of model extras into weights folder, separate download of tokenizer, making engine.py utilize downloaded tokenizer, model and tokenizer revision --- Dockerfile | 6 ++++++ README.md | 12 ++++++++--- src/download_model.py | 50 ++++++++++++++++++++++++++++++++----------- src/engine.py | 15 +++++++++++-- 4 files changed, 65 insertions(+), 18 deletions(-) diff --git a/Dockerfile b/Dockerfile index 65986a3..49b5d0d 100644 --- a/Dockerfile +++ b/Dockerfile @@ -15,10 +15,16 @@ COPY src /src # Setup for Option 2: Building the Image with the Model included ARG MODEL_NAME="" +ARG TOKENIZER_NAME="" ARG BASE_PATH="/runpod-volume" ARG QUANTIZATION="" +ARG MODEL_REVISION="main" +ARG TOKENIZER_REVISION="main" ENV MODEL_NAME=$MODEL_NAME \ + MODEL_REVISION=$REVISION \ + TOKENIZER_NAME=$TOKENIZER_NAME \ + TOKENIZER_REVISION=$TOKENIZER_REVISION \ BASE_PATH=$BASE_PATH \ QUANTIZATION=$QUANTIZATION \ HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \ diff --git a/README.md b/README.md index eaabb6b..e89ef1d 100644 --- a/README.md +++ b/README.md @@ -58,8 +58,7 @@ Development Image: ```runpod/worker-vllm:dev``` **Optional**: - LLM Settings: - - `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`) - - `CUSTOM_CHAT_TEMPLATE`: Custom chat jinja template, read more about Hugging Face chat templates [here](https://huggingface.co/docs/transformers/chat_templating). (default: `None`) + - `MODEL_REVISION`: Model revision to load (default: `main`). - `MAX_MODEL_LENGTH`: Maximum number of tokens for the engine to be able to handle. (default: maximum supported by the model) - `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't) - `LOAD_FORMAT`: Format to load model in (default: `auto`). @@ -67,8 +66,12 @@ Development Image: ```runpod/worker-vllm:dev``` - `QUANTIZATION`: AWQ (`awq`), SqueezeLLM (`squeezellm`) or GPTQ (`gptq`) Quantization. The specified Model Repo must be of a quantized model. (default: `None`) - `TRUST_REMOTE_CODE`: Trust remote code for Hugging Face (default: `0`) -- Tensor Parallelism: +- Tokenizer Settings: + - `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer) + - `TOKENIZER_REVISION`: Tokenizer revision to load (default: `main`). + - `CUSTOM_CHAT_TEMPLATE`: Custom chat jinja template, read more about Hugging Face chat templates [here](https://huggingface.co/docs/transformers/chat_templating). (default: `None`) +- Tensor Parallelism: Note that the more GPUs you split a model's weights accross, the slower it will be due to inter-GPU communication overhead. If you can fit the model on a single GPU, it is recommended to do so. - `TENSOR_PARALLEL_SIZE`: Number of GPUs to shard the model across (default: `1`). - If you are having issues loading your model with Tensor Parallelism, try decreasing `VLLM_CPU_FRACTION` (default: `1`). @@ -96,9 +99,12 @@ To build an image with the model baked in, you must specify the following docker - **Required** - `MODEL_NAME` - **Optional** + - `MODEL_REVISION`: Model revision to load (default: `main`). - `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't. If your intention is to bake the model into the image, you should set this to something like `/models` to make sure there are no issues if you were to accidentally attach network storage.) - `QUANTIZATION` - `WORKER_CUDA_VERSION`: `11.8.0` or `12.1.0` (default: `11.8.0` due to a small amount of workers not having CUDA 12.1 support yet. `12.1.0` is recommended for optimal performance). + - `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer) + - `TOKENIZER_REVISION`: Tokenizer revision to load (default: `main`). For the remaining settings, you may apply them as environment variables when running the container. Supported environment variables are listed in the [Environment Variables](#environment-variables) section. diff --git a/src/download_model.py b/src/download_model.py index 12850f9..02d0e26 100644 --- a/src/download_model.py +++ b/src/download_model.py @@ -1,11 +1,16 @@ import os +import shutil import logging from huggingface_hub import snapshot_download -from vllm.model_executor.weight_utils import prepare_hf_model_weights +from vllm.model_executor.weight_utils import Disabledtqdm, prepare_hf_model_weights if __name__ == "__main__": model = os.getenv("MODEL_NAME") download_dir = os.getenv("HF_HOME") + tokenizer = os.getenv("TOKENIZER_NAME", model) + model_revision = os.getenv("MODEL_REVISION", "main") + tokenizer_revision = os.getenv("TOKENIZER_REVISION", "main") + if not model or not download_dir: raise ValueError(f"Must specify model and download_dir. Model: {model}, download_dir: {download_dir}") @@ -14,22 +19,41 @@ if __name__ == "__main__": logging.info(f"Downloading model {model} to {download_dir}") - hf_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights( + model_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights( model_name_or_path=model, cache_dir=download_dir, - ) - - snapshot_download( - model, - cache_dir=download_dir, - allow_patterns=[ - "*token*", - "config.json", - ] + revision=model_revision, ) - logging.info(f"Finished downloading model {model} to {download_dir}") + model_extras_folder = snapshot_download(model, + allow_patterns=[ + "*token*", + "*.json" + ], + cache_dir=download_dir + "/extras", + tqdm_class=Disabledtqdm, + revision=model_revision) + + # Move extras to hf_folder + for f in os.listdir(model_extras_folder): + shutil.move(model_extras_folder + "/" + f, model_folder + "/" + f) # Wrie hf_folder to file with open("/local_model_path.txt", "w") as f: - f.write(hf_folder) \ No newline at end of file + f.write(model_folder) + + logging.info(f"Finished downloading model {model} to {download_dir}") + + if tokenizer != model: + logging.info(f"Downloading tokenizer {tokenizer} to {download_dir}") + tokenizer_folder = snapshot_download(tokenizer, + cache_dir=download_dir, + tqdm_class=Disabledtqdm, + revision=tokenizer_revision) + with open("/local_tokenizer_path.txt", "w") as f: + f.write(tokenizer_folder) + + logging.info(f"Finished downloading tokenizer {tokenizer} to {download_dir}") + + + diff --git a/src/engine.py b/src/engine.py index 356fee3..dedf370 100644 --- a/src/engine.py +++ b/src/engine.py @@ -41,7 +41,7 @@ class vLLMEngine: load_dotenv() # For local development self.config = self._initialize_config() logging.info("vLLM config: %s", self.config) - self.tokenizer = Tokenizer(os.getenv("TOKENIZER_NAME", os.getenv("MODEL_NAME"))) + self.tokenizer = Tokenizer(self.config["tokenizer_name_or_path"]) self.llm = self._initialize_llm() if engine is None else engine self.openai_engine = self._initialize_openai() self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY)) @@ -166,6 +166,9 @@ class vLLMEngine: def _initialize_config(self): quantization = self._get_quantization() model, download_dir = self._get_model_name_and_path() + tokenizer_name_or_path = self._get_tokenizer_name_or_path() + if not tokenizer_name_or_path: + tokenizer_name_or_path = model return { "model": model, @@ -173,7 +176,7 @@ class vLLMEngine: "quantization": quantization, "load_format": os.getenv("LOAD_FORMAT", "auto"), "dtype": "half" if quantization else "auto", - "tokenizer": os.getenv("TOKENIZER_NAME"), + "tokenizer": tokenizer_name_or_path, "disable_log_stats": bool(int(os.getenv("DISABLE_LOG_STATS", 1))), "disable_log_requests": bool(int(os.getenv("DISABLE_LOG_REQUESTS", 1))), "trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))), @@ -209,6 +212,14 @@ class vLLMEngine: else: model, download_dir = os.getenv("MODEL_NAME"), os.getenv("HF_HOME") return model, download_dir + + def _get_tokenizer_name_or_path(self): + if os.path.exists("/local_tokenizer_path.txt"): + tokenizer_name_or_path = open("/local_tokenizer_path.txt", "r").read().strip() + logging.info("Using local tokenizer at %s", tokenizer_name_or_path) + else: + tokenizer_name_or_path = os.getenv("TOKENIZER_NAME") + return tokenizer_name_or_path def _get_num_gpu_shard(self): num_gpu_shard = int(os.getenv("TENSOR_PARALLEL_SIZE", 1)) From 8de10468dd418afadf161995a053efe92277cac7 Mon Sep 17 00:00:00 2001 From: alpayariyak Date: Sat, 3 Feb 2024 00:28:21 +0000 Subject: [PATCH 6/6] Working tokenizer and model download fix Fix handler startup --- Dockerfile | 18 ++++++------ README.md | 6 ++-- builder/download_model.py | 50 +++++++++++++++++++++++++++++++++ src/download_model.py | 59 --------------------------------------- src/engine.py | 31 ++++++++++---------- 5 files changed, 79 insertions(+), 85 deletions(-) create mode 100644 builder/download_model.py delete mode 100644 src/download_model.py diff --git a/Dockerfile b/Dockerfile index 49b5d0d..0790c31 100644 --- a/Dockerfile +++ b/Dockerfile @@ -10,16 +10,13 @@ RUN --mount=type=cache,target=/root/.cache/pip \ python3 -m pip install --upgrade pip && \ python3 -m pip install --upgrade -r /requirements.txt -# Add source files -COPY src /src - # Setup for Option 2: Building the Image with the Model included ARG MODEL_NAME="" ARG TOKENIZER_NAME="" ARG BASE_PATH="/runpod-volume" ARG QUANTIZATION="" -ARG MODEL_REVISION="main" -ARG TOKENIZER_REVISION="main" +ARG MODEL_REVISION="" +ARG TOKENIZER_REVISION="" ENV MODEL_NAME=$MODEL_NAME \ MODEL_REVISION=$REVISION \ @@ -33,14 +30,19 @@ ENV MODEL_NAME=$MODEL_NAME \ HF_TRANSFER=1 ENV PYTHONPATH="/:/vllm-installation" - + +COPY builder/download_model.py /download_model.py RUN --mount=type=secret,id=HF_TOKEN,required=false \ if [ -f /run/secrets/HF_TOKEN ]; then \ export HF_TOKEN=$(cat /run/secrets/HF_TOKEN); \ fi && \ if [ -n "$MODEL_NAME" ]; then \ - python3 /src/download_model.py; \ + python3 /download_model.py; \ fi +# Add source files +COPY src /src + + # Start the handler -CMD ["python3", "/src/handler.py"] +CMD ["python3", "/src/handler.py"] \ No newline at end of file diff --git a/README.md b/README.md index e89ef1d..2fb7bad 100644 --- a/README.md +++ b/README.md @@ -58,7 +58,7 @@ Development Image: ```runpod/worker-vllm:dev``` **Optional**: - LLM Settings: - - `MODEL_REVISION`: Model revision to load (default: `main`). + - `MODEL_REVISION`: Model revision to load (default: `None`). - `MAX_MODEL_LENGTH`: Maximum number of tokens for the engine to be able to handle. (default: maximum supported by the model) - `BASE_PATH`: Storage directory where huggingface cache and model will be located. (default: `/runpod-volume`, which will utilize network storage if you attach it or create a local directory within the image if you don't) - `LOAD_FORMAT`: Format to load model in (default: `auto`). @@ -68,7 +68,7 @@ Development Image: ```runpod/worker-vllm:dev``` - Tokenizer Settings: - `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer) - - `TOKENIZER_REVISION`: Tokenizer revision to load (default: `main`). + - `TOKENIZER_REVISION`: Tokenizer revision to load (default: `None`). - `CUSTOM_CHAT_TEMPLATE`: Custom chat jinja template, read more about Hugging Face chat templates [here](https://huggingface.co/docs/transformers/chat_templating). (default: `None`) - Tensor Parallelism: @@ -78,7 +78,7 @@ Development Image: ```runpod/worker-vllm:dev``` - System Settings: - `GPU_MEMORY_UTILIZATION`: GPU VRAM utilization (default: `0.98`). - - `MAX_PARALLEL_LOADING_WORKERS`: Maximum number of parallel workers for loading models (default: `number of available CPU cores` if `TENSOR_PARALLEL_SIZE` is `1`, otherwise `None`). + - `MAX_PARALLEL_LOADING_WORKERS`: Maximum number of parallel workers for loading models, for non-Tensor Parallel only. (default: `number of available CPU cores` if `TENSOR_PARALLEL_SIZE` is `1`, otherwise `None`). - Serverless Settings: diff --git a/builder/download_model.py b/builder/download_model.py new file mode 100644 index 0000000..caf4a42 --- /dev/null +++ b/builder/download_model.py @@ -0,0 +1,50 @@ +import os +import shutil +from huggingface_hub import snapshot_download +from vllm.model_executor.weight_utils import prepare_hf_model_weights, Disabledtqdm + +def download_extras_or_tokenizer(model_name, cache_dir, revision, extras=False): + """Download model or tokenizer and prepare its weights, returning the local folder path.""" + pattern = ["*token*", "*.json"] if extras else None + extra_dir = "/extras" if extras else "" + folder = snapshot_download( + model_name, + cache_dir=cache_dir + extra_dir, + revision=revision, + tqdm_class=Disabledtqdm, + allow_patterns=pattern if extras else None, + ignore_patterns=["*.safetensors", "*.bin", "*.pt"] if not extras else None + ) + return folder + +def move_files(src_dir, dest_dir): + """Move files from source to destination directory.""" + for f in os.listdir(src_dir): + src_path = os.path.join(src_dir, f) + dst_path = os.path.join(dest_dir, f) + shutil.copy2(src_path, dst_path) + os.remove(src_path) + +if __name__ == "__main__": + model, download_dir = os.getenv("MODEL_NAME"), os.getenv("HF_HOME") + tokenizer = os.getenv("TOKENIZER_NAME", model) + revisions = { + "model": os.getenv("MODEL_REVISION") or None, + "tokenizer": os.getenv("TOKENIZER_REVISION") or None + } + + if not model or not download_dir: + raise ValueError(f"Must specify model and download_dir. Model: {model}, download_dir: {download_dir}") + + os.makedirs(download_dir, exist_ok=True) + model_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights(model_name_or_path=model, revision=revisions["model"], cache_dir=download_dir) + model_extras_folder = download_extras_or_tokenizer(model, download_dir, revisions["model"], extras=True) + move_files(model_extras_folder, model_folder) + + with open("/local_model_path.txt", "w") as f: + f.write(model_folder) + + if tokenizer != model: + tokenizer_folder = download_extras_or_tokenizer(tokenizer, download_dir, revisions["tokenizer"]) + with open("/local_tokenizer_path.txt", "w") as f: + f.write(tokenizer_folder) diff --git a/src/download_model.py b/src/download_model.py deleted file mode 100644 index 02d0e26..0000000 --- a/src/download_model.py +++ /dev/null @@ -1,59 +0,0 @@ -import os -import shutil -import logging -from huggingface_hub import snapshot_download -from vllm.model_executor.weight_utils import Disabledtqdm, prepare_hf_model_weights - -if __name__ == "__main__": - model = os.getenv("MODEL_NAME") - download_dir = os.getenv("HF_HOME") - tokenizer = os.getenv("TOKENIZER_NAME", model) - model_revision = os.getenv("MODEL_REVISION", "main") - tokenizer_revision = os.getenv("TOKENIZER_REVISION", "main") - - if not model or not download_dir: - raise ValueError(f"Must specify model and download_dir. Model: {model}, download_dir: {download_dir}") - - if not os.path.exists(download_dir): - os.makedirs(download_dir) - - logging.info(f"Downloading model {model} to {download_dir}") - - model_folder, hf_weights_files, use_safetensors = prepare_hf_model_weights( - model_name_or_path=model, - cache_dir=download_dir, - revision=model_revision, - ) - - model_extras_folder = snapshot_download(model, - allow_patterns=[ - "*token*", - "*.json" - ], - cache_dir=download_dir + "/extras", - tqdm_class=Disabledtqdm, - revision=model_revision) - - # Move extras to hf_folder - for f in os.listdir(model_extras_folder): - shutil.move(model_extras_folder + "/" + f, model_folder + "/" + f) - - # Wrie hf_folder to file - with open("/local_model_path.txt", "w") as f: - f.write(model_folder) - - logging.info(f"Finished downloading model {model} to {download_dir}") - - if tokenizer != model: - logging.info(f"Downloading tokenizer {tokenizer} to {download_dir}") - tokenizer_folder = snapshot_download(tokenizer, - cache_dir=download_dir, - tqdm_class=Disabledtqdm, - revision=tokenizer_revision) - with open("/local_tokenizer_path.txt", "w") as f: - f.write(tokenizer_folder) - - logging.info(f"Finished downloading tokenizer {tokenizer} to {download_dir}") - - - diff --git a/src/engine.py b/src/engine.py index dedf370..b165943 100644 --- a/src/engine.py +++ b/src/engine.py @@ -13,8 +13,8 @@ from dotenv import load_dotenv class Tokenizer: - def __init__(self, model_name): - self.tokenizer = AutoTokenizer.from_pretrained(model_name) + def __init__(self, tokenizer_name_or_path, tokenizer_revision): + self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_name_or_path, revision=tokenizer_revision) self.custom_chat_template = os.getenv("CUSTOM_CHAT_TEMPLATE") self.has_chat_template = bool(self.tokenizer.chat_template) or bool(self.custom_chat_template) if self.custom_chat_template and isinstance(self.custom_chat_template, str): @@ -41,7 +41,7 @@ class vLLMEngine: load_dotenv() # For local development self.config = self._initialize_config() logging.info("vLLM config: %s", self.config) - self.tokenizer = Tokenizer(self.config["tokenizer_name_or_path"]) + self.tokenizer = Tokenizer(self.config["tokenizer"], self.config["tokenizer_revision"]) self.llm = self._initialize_llm() if engine is None else engine self.openai_engine = self._initialize_openai() self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY)) @@ -60,7 +60,6 @@ class vLLMEngine: yield batch async def generate_vllm(self, llm_input, validated_sampling_params, batch_size, stream, apply_chat_template, request_id: str) -> AsyncGenerator[dict, None]: - if apply_chat_template or isinstance(llm_input, list): llm_input = self.tokenizer.apply_chat_template(llm_input) validated_sampling_params = SamplingParams(**validated_sampling_params) @@ -165,18 +164,20 @@ class vLLMEngine: def _initialize_config(self): quantization = self._get_quantization() - model, download_dir = self._get_model_name_and_path() - tokenizer_name_or_path = self._get_tokenizer_name_or_path() + model, download_dir, model_revision = self._get_model_info() + tokenizer_name_or_path, tokenizer_revision = self._get_tokenizer_info() if not tokenizer_name_or_path: tokenizer_name_or_path = model return { "model": model, + "revision": model_revision, "download_dir": download_dir, "quantization": quantization, "load_format": os.getenv("LOAD_FORMAT", "auto"), "dtype": "half" if quantization else "auto", "tokenizer": tokenizer_name_or_path, + "tokenizer_revision": tokenizer_revision, "disable_log_stats": bool(int(os.getenv("DISABLE_LOG_STATS", 1))), "disable_log_requests": bool(int(os.getenv("DISABLE_LOG_REQUESTS", 1))), "trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))), @@ -204,22 +205,22 @@ class vLLMEngine: return None else: return int(os.getenv("MAX_PARALLEL_LOADING_WORKERS", count_physical_cores())) - - def _get_model_name_and_path(self): + + def _get_model_info(self): if os.path.exists("/local_model_path.txt"): - model, download_dir = open("/local_model_path.txt", "r").read().strip(), None + model, download_dir, revision = open("/local_model_path.txt", "r").read().strip(), None, None logging.info("Using local model at %s", model) else: - model, download_dir = os.getenv("MODEL_NAME"), os.getenv("HF_HOME") - return model, download_dir + model, download_dir, revision = os.getenv("MODEL_NAME"), os.getenv("HF_HOME"), os.getenv("MODEL_REVISION") or None + return model, download_dir, revision - def _get_tokenizer_name_or_path(self): + def _get_tokenizer_info(self): if os.path.exists("/local_tokenizer_path.txt"): - tokenizer_name_or_path = open("/local_tokenizer_path.txt", "r").read().strip() + tokenizer_name_or_path, revision = open("/local_tokenizer_path.txt", "r").read().strip(), None logging.info("Using local tokenizer at %s", tokenizer_name_or_path) else: - tokenizer_name_or_path = os.getenv("TOKENIZER_NAME") - return tokenizer_name_or_path + tokenizer_name_or_path, revision = os.getenv("TOKENIZER_NAME"), os.getenv("TOKENIZER_REVISION") or None + return tokenizer_name_or_path, revision def _get_num_gpu_shard(self): num_gpu_shard = int(os.getenv("TENSOR_PARALLEL_SIZE", 1))