Complete refactor, improved overall functionality
This commit is contained in:
+33
-32
@@ -1,53 +1,54 @@
|
||||
# Base image - Set default to CUDA 11.8.0
|
||||
ARG CUDA_VERSION=11.8.0
|
||||
# Base image - Set default to CUDA 11.8
|
||||
ARG WORKER_CUDA_VERSION=11.8
|
||||
FROM runpod/base:0.4.2-cuda${WORKER_CUDA_VERSION}.0 as builder
|
||||
|
||||
# Use different base images based on CUDA_VERSION argument
|
||||
FROM runpod/base:0.4.2-cuda${CUDA_VERSION} as builder
|
||||
ARG WORKER_CUDA_VERSION=11.8 # Required duplicate to keep in scope
|
||||
|
||||
ENV HF_DATASETS_CACHE="/runpod-volume/huggingface-cache/datasets" \
|
||||
# Set Environment Variables
|
||||
ENV WORKER_CUDA_VERSION=${WORKER_CUDA_VERSION} \
|
||||
HF_DATASETS_CACHE="/runpod-volume/huggingface-cache/datasets" \
|
||||
HUGGINGFACE_HUB_CACHE="/runpod-volume/huggingface-cache/hub" \
|
||||
TRANSFORMERS_CACHE="/runpod-volume/huggingface-cache/hub"
|
||||
TRANSFORMERS_CACHE="/runpod-volume/huggingface-cache/hub"
|
||||
|
||||
# Install Python dependencies (Worker Template)
|
||||
|
||||
# Install Python dependencies
|
||||
COPY builder/requirements.txt /requirements.txt
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
python3.11 -m pip install --upgrade pip && \
|
||||
python3.11 -m pip install --upgrade -r /requirements.txt --no-cache-dir && \
|
||||
python3.11 -m pip install --upgrade -r /requirements.txt && \
|
||||
rm /requirements.txt
|
||||
|
||||
# Install specific packages based on CUDA version
|
||||
# RUN if [ "$CUDA_VERSION" == "12.1.0" ]; then \
|
||||
# python3.11 -m pip install vllm==0.2.3; \
|
||||
# else \
|
||||
# python3.11 -m pip install https://github.com/vllm-project/vllm/releases/download/v0.2.4/vllm-0.2.4+cu118-cp311-cp311-manylinux1_x86_64.whl; \
|
||||
# fi
|
||||
# Install torch and vllm based on CUDA version
|
||||
RUN if [[ "${WORKER_CUDA_VERSION}" == 11.8* ]]; then \
|
||||
wget https://github.com/alpayariyak/vllm/releases/download/0.2.4-runpod-11.8/vllm-0.2.4+cu118-cp311-cp311-linux_x86_64.whl && \
|
||||
python3.11 -m pip install vllm-0.2.4+cu118-cp311-cp311-linux_x86_64.whl && \
|
||||
rm vllm-0.2.4+cu118-cp311-cp311-linux_x86_64.whl; \
|
||||
python3.11 -m pip uninstall torch -y; \
|
||||
python3.11 -m pip install torch --upgrade --index-url https://download.pytorch.org/whl/cu118; \
|
||||
python3.11 -m pip uninstall xformers -y; \
|
||||
python3.11 -m pip install --upgrade xformers --index-url https://download.pytorch.org/whl/cu118; \
|
||||
else \
|
||||
python3.11 -m pip install -e git+https://github.com/alpayariyak/vllm.git#egg=vllm; \
|
||||
fi && \
|
||||
rm -rf /root/.cache/pip
|
||||
|
||||
|
||||
# Add source files
|
||||
ADD src .
|
||||
COPY src .
|
||||
|
||||
# Setup for Option 2: Building the Image with the Model included
|
||||
ARG MODEL_NAME=""
|
||||
ARG MODEL_BASE_PATH="/runpod-volume/"
|
||||
ARG HF_TOKEN=""
|
||||
ARG QUANTIZATION=""
|
||||
|
||||
# Conditionally run download_model.py
|
||||
RUN if [ -n "$MODEL_NAME" ]; then \
|
||||
export HF_TOKEN=$HF_TOKEN; \
|
||||
python3.11 /download_model.py --model $MODEL_NAME --download_dir $MODEL_BASE_PATH; \
|
||||
export MODEL_NAME=$MODEL_NAME; \
|
||||
export MODEL_BASE_PATH=$MODEL_BASE_PATH; \
|
||||
fi
|
||||
|
||||
RUN if [ -n "$QUANTIZATION" ]; then \
|
||||
python3.11 /download_model.py --model $MODEL_NAME --download_dir $MODEL_BASE_PATH; \
|
||||
export MODEL_BASE_PATH=$MODEL_BASE_PATH; \
|
||||
export MODEL_NAME=$MODEL_NAME; \
|
||||
fi && \
|
||||
if [ -n "$QUANTIZATION" ]; then \
|
||||
export QUANTIZATION=$QUANTIZATION; \
|
||||
fi
|
||||
|
||||
RUN mkdir inference_engine && \
|
||||
cd inference_engine && \
|
||||
git clone https://github.com/alpayariyak/vllm.git && \
|
||||
cd vllm && \
|
||||
python3.11 -m pip install -e . && \
|
||||
cd ../..
|
||||
|
||||
# Start the handler
|
||||
CMD ["python3.11", "/handler.py"]
|
||||
CMD ["python3.11", "/handler.py"]
|
||||
Reference in New Issue
Block a user