diff --git a/Dockerfile b/Dockerfile index 5105496..5ea7240 100644 --- a/Dockerfile +++ b/Dockerfile @@ -46,5 +46,11 @@ RUN --mount=type=secret,id=HF_TOKEN,required=false \ python3 /src/download_model.py; \ fi + +# Create directory for startup script and copy it +RUN mkdir -p /usr/local/bin +COPY --chmod=755 start.sh /usr/local/bin/start.sh + # Start the handler +ENTRYPOINT ["/usr/local/bin/start.sh"] CMD ["python3", "/src/handler.py"] diff --git a/start.sh b/start.sh new file mode 100644 index 0000000..64afddd --- /dev/null +++ b/start.sh @@ -0,0 +1,28 @@ +# /usr/local/bin/start.sh +#!/usr/bin/env bash +set -euo pipefail + +# If user didn’t set it explicitly, infer from visible GPUs. +if [[ -z "${TENSOR_PARALLEL_SIZE:-}" ]]; then + if command -v nvidia-smi >/dev/null 2>&1; then + COUNT="$(nvidia-smi -L | wc -l | tr -d ' ')" + else + # Fallback to PyTorch if available + COUNT="$(python3 - <<'PY' +try: + import torch + print(torch.cuda.device_count() or 0) +except Exception: + print(0) +PY +)" + fi + + # Respect CUDA_VISIBLE_DEVICES (both methods above do, since they see only visible GPUs). + if [[ "${COUNT}" -lt 1 ]]; then + COUNT=1 + fi + export TENSOR_PARALLEL_SIZE="${COUNT}" +fi + +echo "TENSOR_PARALLEL_SIZE=${TENSOR_PARALLEL_SIZE}"