diff --git a/Dockerfile b/Dockerfile index 94a5456..fd5f2c2 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,7 +1,5 @@ # Base image # The following docker base image is recommended by VLLM: -# FROM runpod/pytorch:2.0.1-py3.10-cuda11.8.0-devel -# FROM nvcr.io/nvidia/pytorch:22.12-py3 FROM runpod/pytorch:2.0.1-py3.10-cuda11.8.0-devel # Use bash shell with pipefail option @@ -19,6 +17,9 @@ RUN chmod +x /setup.sh && \ /setup.sh && \ rm /setup.sh +# Install fast api +RUN pip install fastapi==0.99.1 + # Install Python dependencies (Worker Template) COPY builder/requirements.txt /requirements.txt RUN --mount=type=cache,target=/root/.cache/pip \ @@ -30,13 +31,17 @@ RUN --mount=type=cache,target=/root/.cache/pip \ ADD src . # Quick temporary updates -RUN pip install git+https://github.com/runpod/runpod-python@multijob2#egg=runpod --compile +RUN pip install git+https://github.com/runpod/runpod-python@main#egg=runpod --compile # Prepare the models inside the docker image ARG HUGGING_FACE_HUB_TOKEN=NONE ENV HUGGING_FACE_HUB_TOKEN=$HUGGING_FACE_HUB_TOKEN -ENV DOWNLOAD_7B_MODEL=YES -# ENV DOWNLOAD_13B_MODEL=1 + +ARG DOWNLOAD_7B_MODEL= +ENV DOWNLOAD_7B_MODEL=$DOWNLOAD_7B_MODEL + +#ARG DOWNLOAD_13B_MODEL= +#ENV DOWNLOAD_13B_MODEL=$DOWNLOAD_13B_MODEL # Download the models RUN mkdir -p /model diff --git a/builder/requirements.txt b/builder/requirements.txt index c83ebc2..a4d379b 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -7,4 +7,4 @@ # runpod @ git+https://github.com/runpod/runpod-python@vllm#egg=runpod vllm==0.1.2 huggingface-hub==0.16.4 -runpod @ git+https://github.com/runpod/runpod-python@multijob2#egg=runpod +runpod @ git+https://github.com/runpod/runpod-python@main#egg=runpod diff --git a/src/handler.py b/src/handler.py index 9eb680c..54b03de 100644 --- a/src/handler.py +++ b/src/handler.py @@ -24,7 +24,7 @@ engine_args = AsyncEngineArgs( # Create the vLLM asynchronous engine llm = AsyncLLMEngine.from_engine_args(engine_args) -def handler_fully_utilized() -> bool: +def concurrency_controller() -> bool: # Compute pending sequences total_pending_sequences = len(llm.engine.scheduler.waiting) + len(llm.engine.scheduler.swapped) return total_pending_sequences > 10 @@ -88,7 +88,11 @@ async def handler(job): job_input = job['input'] # Prompts - prompt = job_input['prompt'] + template = """SYSTEM: You are a helpful assistant. +USER: {} +ASSISTANT: """ + + prompt = template.format(job_input['prompt']) # Streaming streaming = job_input.get('streaming', False) @@ -142,4 +146,4 @@ async def handler(job): else: return await submit_output() -runpod.serverless.start({"handler": handler, "handler_fully_utilized": handler_fully_utilized}) +runpod.serverless.start({"handler": handler, "concurrency_controller": concurrency_controller})