This commit is contained in:
Jorg Doku
2023-07-28 23:14:03 -05:00
parent 67b6cb8b33
commit c862d043dc
3 changed files with 18 additions and 9 deletions
+10 -5
View File
@@ -1,7 +1,5 @@
# Base image
# The following docker base image is recommended by VLLM:
# FROM runpod/pytorch:2.0.1-py3.10-cuda11.8.0-devel
# FROM nvcr.io/nvidia/pytorch:22.12-py3
FROM runpod/pytorch:2.0.1-py3.10-cuda11.8.0-devel
# Use bash shell with pipefail option
@@ -19,6 +17,9 @@ RUN chmod +x /setup.sh && \
/setup.sh && \
rm /setup.sh
# Install fast api
RUN pip install fastapi==0.99.1
# Install Python dependencies (Worker Template)
COPY builder/requirements.txt /requirements.txt
RUN --mount=type=cache,target=/root/.cache/pip \
@@ -30,13 +31,17 @@ RUN --mount=type=cache,target=/root/.cache/pip \
ADD src .
# Quick temporary updates
RUN pip install git+https://github.com/runpod/runpod-python@multijob2#egg=runpod --compile
RUN pip install git+https://github.com/runpod/runpod-python@main#egg=runpod --compile
# Prepare the models inside the docker image
ARG HUGGING_FACE_HUB_TOKEN=NONE
ENV HUGGING_FACE_HUB_TOKEN=$HUGGING_FACE_HUB_TOKEN
ENV DOWNLOAD_7B_MODEL=YES
# ENV DOWNLOAD_13B_MODEL=1
ARG DOWNLOAD_7B_MODEL=
ENV DOWNLOAD_7B_MODEL=$DOWNLOAD_7B_MODEL
#ARG DOWNLOAD_13B_MODEL=
#ENV DOWNLOAD_13B_MODEL=$DOWNLOAD_13B_MODEL
# Download the models
RUN mkdir -p /model
+1 -1
View File
@@ -7,4 +7,4 @@
# runpod @ git+https://github.com/runpod/runpod-python@vllm#egg=runpod
vllm==0.1.2
huggingface-hub==0.16.4
runpod @ git+https://github.com/runpod/runpod-python@multijob2#egg=runpod
runpod @ git+https://github.com/runpod/runpod-python@main#egg=runpod
+7 -3
View File
@@ -24,7 +24,7 @@ engine_args = AsyncEngineArgs(
# Create the vLLM asynchronous engine
llm = AsyncLLMEngine.from_engine_args(engine_args)
def handler_fully_utilized() -> bool:
def concurrency_controller() -> bool:
# Compute pending sequences
total_pending_sequences = len(llm.engine.scheduler.waiting) + len(llm.engine.scheduler.swapped)
return total_pending_sequences > 10
@@ -88,7 +88,11 @@ async def handler(job):
job_input = job['input']
# Prompts
prompt = job_input['prompt']
template = """SYSTEM: You are a helpful assistant.
USER: {}
ASSISTANT: """
prompt = template.format(job_input['prompt'])
# Streaming
streaming = job_input.get('streaming', False)
@@ -142,4 +146,4 @@ async def handler(job):
else:
return await submit_output()
runpod.serverless.start({"handler": handler, "handler_fully_utilized": handler_fully_utilized})
runpod.serverless.start({"handler": handler, "concurrency_controller": concurrency_controller})