latest
This commit is contained in:
+10
-5
@@ -1,7 +1,5 @@
|
||||
# Base image
|
||||
# The following docker base image is recommended by VLLM:
|
||||
# FROM runpod/pytorch:2.0.1-py3.10-cuda11.8.0-devel
|
||||
# FROM nvcr.io/nvidia/pytorch:22.12-py3
|
||||
FROM runpod/pytorch:2.0.1-py3.10-cuda11.8.0-devel
|
||||
|
||||
# Use bash shell with pipefail option
|
||||
@@ -19,6 +17,9 @@ RUN chmod +x /setup.sh && \
|
||||
/setup.sh && \
|
||||
rm /setup.sh
|
||||
|
||||
# Install fast api
|
||||
RUN pip install fastapi==0.99.1
|
||||
|
||||
# Install Python dependencies (Worker Template)
|
||||
COPY builder/requirements.txt /requirements.txt
|
||||
RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
@@ -30,13 +31,17 @@ RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
ADD src .
|
||||
|
||||
# Quick temporary updates
|
||||
RUN pip install git+https://github.com/runpod/runpod-python@multijob2#egg=runpod --compile
|
||||
RUN pip install git+https://github.com/runpod/runpod-python@main#egg=runpod --compile
|
||||
|
||||
# Prepare the models inside the docker image
|
||||
ARG HUGGING_FACE_HUB_TOKEN=NONE
|
||||
ENV HUGGING_FACE_HUB_TOKEN=$HUGGING_FACE_HUB_TOKEN
|
||||
ENV DOWNLOAD_7B_MODEL=YES
|
||||
# ENV DOWNLOAD_13B_MODEL=1
|
||||
|
||||
ARG DOWNLOAD_7B_MODEL=
|
||||
ENV DOWNLOAD_7B_MODEL=$DOWNLOAD_7B_MODEL
|
||||
|
||||
#ARG DOWNLOAD_13B_MODEL=
|
||||
#ENV DOWNLOAD_13B_MODEL=$DOWNLOAD_13B_MODEL
|
||||
|
||||
# Download the models
|
||||
RUN mkdir -p /model
|
||||
|
||||
@@ -7,4 +7,4 @@
|
||||
# runpod @ git+https://github.com/runpod/runpod-python@vllm#egg=runpod
|
||||
vllm==0.1.2
|
||||
huggingface-hub==0.16.4
|
||||
runpod @ git+https://github.com/runpod/runpod-python@multijob2#egg=runpod
|
||||
runpod @ git+https://github.com/runpod/runpod-python@main#egg=runpod
|
||||
|
||||
+7
-3
@@ -24,7 +24,7 @@ engine_args = AsyncEngineArgs(
|
||||
# Create the vLLM asynchronous engine
|
||||
llm = AsyncLLMEngine.from_engine_args(engine_args)
|
||||
|
||||
def handler_fully_utilized() -> bool:
|
||||
def concurrency_controller() -> bool:
|
||||
# Compute pending sequences
|
||||
total_pending_sequences = len(llm.engine.scheduler.waiting) + len(llm.engine.scheduler.swapped)
|
||||
return total_pending_sequences > 10
|
||||
@@ -88,7 +88,11 @@ async def handler(job):
|
||||
job_input = job['input']
|
||||
|
||||
# Prompts
|
||||
prompt = job_input['prompt']
|
||||
template = """SYSTEM: You are a helpful assistant.
|
||||
USER: {}
|
||||
ASSISTANT: """
|
||||
|
||||
prompt = template.format(job_input['prompt'])
|
||||
|
||||
# Streaming
|
||||
streaming = job_input.get('streaming', False)
|
||||
@@ -142,4 +146,4 @@ async def handler(job):
|
||||
else:
|
||||
return await submit_output()
|
||||
|
||||
runpod.serverless.start({"handler": handler, "handler_fully_utilized": handler_fully_utilized})
|
||||
runpod.serverless.start({"handler": handler, "concurrency_controller": concurrency_controller})
|
||||
|
||||
Reference in New Issue
Block a user