diff --git a/Dockerfile b/Dockerfile index 5105496..454a085 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,9 +1,9 @@ -FROM nvidia/cuda:12.1.0-base-ubuntu22.04 +FROM nvidia/cuda:12.4.1-base-ubuntu22.04 RUN apt-get update -y \ && apt-get install -y python3-pip -RUN ldconfig /usr/local/cuda-12.1/compat/ +RUN ldconfig /usr/local/cuda-12.4/compat/ # Install Python dependencies COPY builder/requirements.txt /requirements.txt @@ -11,9 +11,8 @@ RUN --mount=type=cache,target=/root/.cache/pip \ python3 -m pip install --upgrade pip && \ python3 -m pip install --upgrade -r /requirements.txt -# Install vLLM (switching back to pip installs since issues that required building fork are fixed and space optimization is not as important since caching) and FlashInfer -RUN python3 -m pip install vllm==0.11.0 && \ - python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3 +# Install vLLM +RUN python3 -m pip install vllm==0.11.0 # Setup for Option 2: Building the Image with the Model included ARG MODEL_NAME=""