diff --git a/Dockerfile b/Dockerfile index b77db07..5105496 100644 --- a/Dockerfile +++ b/Dockerfile @@ -46,5 +46,5 @@ RUN --mount=type=secret,id=HF_TOKEN,required=false \ python3 /src/download_model.py; \ fi - +# Start the handler CMD ["python3", "/src/handler.py"] diff --git a/src/handler.py b/src/handler.py index 4e6982c..176ec7e 100644 --- a/src/handler.py +++ b/src/handler.py @@ -3,20 +3,6 @@ import runpod from utils import JobInput from engine import vLLMEngine, OpenAIvLLMEngine -# Detect number of visible GPUs -gpu_count = torch.cuda.device_count() - -# Fallback to 1 if none detected -if gpu_count < 1: - gpu_count = 1 - -# Set the environment variable -os.environ["TENSOR_PARALLEL_SIZE"] = str(gpu_count) - -print(f"Detected {gpu_count} GPU(s)") -print(f"Set TENSOR_PARALLEL_SIZE={os.environ['TENSOR_PARALLEL_SIZE']}") - - vllm_engine = vLLMEngine() OpenAIvLLMEngine = OpenAIvLLMEngine(vllm_engine)