From 5e38a1a2fa559eb697ce32101943d0200fc6e0a8 Mon Sep 17 00:00:00 2001 From: alpayariyak Date: Sat, 9 Dec 2023 10:36:26 +0000 Subject: [PATCH] Multi-GPU Fix --- Dockerfile | 13 ++++++++++--- README.md | 19 ++++++++++--------- builder/requirements.txt | 2 +- 3 files changed, 21 insertions(+), 13 deletions(-) diff --git a/Dockerfile b/Dockerfile index 82bcc76..74c97ca 100644 --- a/Dockerfile +++ b/Dockerfile @@ -27,16 +27,23 @@ RUN if [ "$CUDA_VERSION" == "12.1.0" ]; then \ ADD src . ARG MODEL_NAME="" -ARG MODEL_BASE_PATH="" +ARG MODEL_BASE_PATH="/runpod-volume/" ARG HF_TOKEN="" -ENV HF_TOKEN=$HF_TOKEN +ARG QUANTIZATION="" # Conditionally run download_model.py -RUN if [ -n "$MODEL_NAME" ] && [ -n "$MODEL_BASE_PATH" ]; then \ +RUN if [ -n "$MODEL_NAME" ]; then \ + export HF_TOKEN=$HF_TOKEN; \ python3.11 /download_model.py --model $MODEL_NAME --download_dir $MODEL_BASE_PATH; \ export MODEL_NAME=$MODEL_NAME; \ export MODEL_BASE_PATH=$MODEL_BASE_PATH; \ fi +RUN if [ -n "$QUANTIZATION" ]; then \ + export QUANTIZATION=$QUANTIZATION; \ + fi + +RUN python3.11 -m pip install pydantic==1.10.13 + # Start the handler CMD ["python3.11", "/handler.py"] diff --git a/README.md b/README.md index 06dbf8b..b2528b4 100644 --- a/README.md +++ b/README.md @@ -10,17 +10,18 @@ ## Setting up the Serverless Worker -### Docker Arguments -#### Required: -- `MODEL_NAME`: The Hugging Face model to use. -- `STREAMING`: Whether to use HTTP Streaming or not. -More information on receiving streaming responses from Serverless Endpoints can be found at [Endpoint URLs](https://docs.runpod.io/docs/serverless-endpoint-urls#streamjob_id), and a detailed example at [Llama2 7B Chat | Streaming Token Outputs](https://docs.runpod.io/reference/llama2-7b-chat#streaming-token-outputs). -#### Optional: -- `TOKENIZER`: The specified tokenizer to use. If you want to use the default tokenizer for the model, do not provide this docker argument at all. + +### Option 1: Pre-Built Image + + +### Option 2: Build Image with Model Inside +#### Docker Arguments: +- `MODEL_NAME`: the Hugging Face model to use. +- `MODEL_BASE_PATH`: directory to store the model in +- `HUGGING_FACE_HUB_TOKEN`: Your Hugging Face token to access private or gated models. You can get your token [here](https://huggingface.co/settings/token). - `QUANTIZATION`: `awq` to use AWQ Quantization (Base model must be in AWQ format). `squeezellm` for SqueezeLLM quantization - preliminary support. ### Environment Variables: -#### Optional: -- `HUGGING_FACE_HUB_TOKEN`: Your Hugging Face token to access private or gated models. You can get your token [here](https://huggingface.co/settings/token). + ### Compatible Models - LLaMA & LLaMA-2 - Mistral diff --git a/builder/requirements.txt b/builder/requirements.txt index d4ea3cb..27e48e7 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -1,3 +1,3 @@ hf_transfer runpod==1.4.0 -huggingface-hub==0.19.4 +huggingface-hub