Complete refactor, improved overall functionality

This commit is contained in:
alpayariyak
2023-12-13 10:33:03 +00:00
parent ef89868570
commit 22de0b322d
5 changed files with 178 additions and 178 deletions
+33 -32
View File
@@ -1,53 +1,54 @@
# Base image - Set default to CUDA 11.8.0
ARG CUDA_VERSION=11.8.0
# Base image - Set default to CUDA 11.8
ARG WORKER_CUDA_VERSION=11.8
FROM runpod/base:0.4.2-cuda${WORKER_CUDA_VERSION}.0 as builder
# Use different base images based on CUDA_VERSION argument
FROM runpod/base:0.4.2-cuda${CUDA_VERSION} as builder
ARG WORKER_CUDA_VERSION=11.8 # Required duplicate to keep in scope
ENV HF_DATASETS_CACHE="/runpod-volume/huggingface-cache/datasets" \
# Set Environment Variables
ENV WORKER_CUDA_VERSION=${WORKER_CUDA_VERSION} \
HF_DATASETS_CACHE="/runpod-volume/huggingface-cache/datasets" \
HUGGINGFACE_HUB_CACHE="/runpod-volume/huggingface-cache/hub" \
TRANSFORMERS_CACHE="/runpod-volume/huggingface-cache/hub"
TRANSFORMERS_CACHE="/runpod-volume/huggingface-cache/hub"
# Install Python dependencies (Worker Template)
# Install Python dependencies
COPY builder/requirements.txt /requirements.txt
RUN --mount=type=cache,target=/root/.cache/pip \
python3.11 -m pip install --upgrade pip && \
python3.11 -m pip install --upgrade -r /requirements.txt --no-cache-dir && \
python3.11 -m pip install --upgrade -r /requirements.txt && \
rm /requirements.txt
# Install specific packages based on CUDA version
# RUN if [ "$CUDA_VERSION" == "12.1.0" ]; then \
# python3.11 -m pip install vllm==0.2.3; \
# else \
# python3.11 -m pip install https://github.com/vllm-project/vllm/releases/download/v0.2.4/vllm-0.2.4+cu118-cp311-cp311-manylinux1_x86_64.whl; \
# fi
# Install torch and vllm based on CUDA version
RUN if [[ "${WORKER_CUDA_VERSION}" == 11.8* ]]; then \
wget https://github.com/alpayariyak/vllm/releases/download/0.2.4-runpod-11.8/vllm-0.2.4+cu118-cp311-cp311-linux_x86_64.whl && \
python3.11 -m pip install vllm-0.2.4+cu118-cp311-cp311-linux_x86_64.whl && \
rm vllm-0.2.4+cu118-cp311-cp311-linux_x86_64.whl; \
python3.11 -m pip uninstall torch -y; \
python3.11 -m pip install torch --upgrade --index-url https://download.pytorch.org/whl/cu118; \
python3.11 -m pip uninstall xformers -y; \
python3.11 -m pip install --upgrade xformers --index-url https://download.pytorch.org/whl/cu118; \
else \
python3.11 -m pip install -e git+https://github.com/alpayariyak/vllm.git#egg=vllm; \
fi && \
rm -rf /root/.cache/pip
# Add source files
ADD src .
COPY src .
# Setup for Option 2: Building the Image with the Model included
ARG MODEL_NAME=""
ARG MODEL_BASE_PATH="/runpod-volume/"
ARG HF_TOKEN=""
ARG QUANTIZATION=""
# Conditionally run download_model.py
RUN if [ -n "$MODEL_NAME" ]; then \
export HF_TOKEN=$HF_TOKEN; \
python3.11 /download_model.py --model $MODEL_NAME --download_dir $MODEL_BASE_PATH; \
export MODEL_NAME=$MODEL_NAME; \
export MODEL_BASE_PATH=$MODEL_BASE_PATH; \
fi
RUN if [ -n "$QUANTIZATION" ]; then \
python3.11 /download_model.py --model $MODEL_NAME --download_dir $MODEL_BASE_PATH; \
export MODEL_BASE_PATH=$MODEL_BASE_PATH; \
export MODEL_NAME=$MODEL_NAME; \
fi && \
if [ -n "$QUANTIZATION" ]; then \
export QUANTIZATION=$QUANTIZATION; \
fi
RUN mkdir inference_engine && \
cd inference_engine && \
git clone https://github.com/alpayariyak/vllm.git && \
cd vllm && \
python3.11 -m pip install -e . && \
cd ../..
# Start the handler
CMD ["python3.11", "/handler.py"]
CMD ["python3.11", "/handler.py"]