4 Commits
3 changed files with 14 additions and 5 deletions
+1 -1
View File
@@ -33,7 +33,7 @@ WORKDIR /work
ADD ./src /work
# Install runpod and its dependencies
RUN pip install -r requirements.txt && chmod +x /work/start.sh
RUN pip install -r ./requirements.txt && chmod +x /work/start.sh
# Set the entrypoint
ENTRYPOINT ["/bin/sh", "-c", "/work/start.sh"]
+6
View File
@@ -0,0 +1,6 @@
"""
This is an empty file used to mark the repository as a Runpod-compatible
serverless endpoint because they won't stop pretending it's not.
I'm tired.
"""
+7 -4
View File
@@ -1,5 +1,8 @@
#!/bin/bash
# fail on error:
set -e
# This script starts the llama-server with the command line arguments
# specified in the environment variable LLAMA_SERVER_CMD_ARGS, ensuring
# that the server listens on port 3098. It also starts the handler.py
@@ -16,8 +19,8 @@ if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then
echo "Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace."
fi
# check if the substring -port is in LLAMA_SERVER_CMD_ARGS
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"-port"* ]]; then
# check if the substring -port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error:
if [[ "$LLAMA_SERVER_CMD_ARGS" == *"-port"* ]]; then
echo "Error: You must not define -port in LLAMA_SERVER_CMD_ARGS, as port 3098 is required."
exit 1
fi
@@ -26,13 +29,13 @@ fi
trap cleanup SIGINT SIGTERM
# kill any existing llama-server processes
pgrep llama-server | xargs kill
pkill llama-server
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
# it contains a.e. "-hf modelname -ctx_size 4096".
# We need to pass these arguments to llama-server verbatim.
llama-server $LLAMA_SERVER_CMD_ARGS -port 3098 2>&1 | tee llama.server.log &
/app/llama-server $LLAMA_SERVER_CMD_ARGS -port 3098 2>&1 | tee llama.server.log &
LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command