5 Commits
6 changed files with 39 additions and 18 deletions
+1 -1
View File
@@ -17,7 +17,7 @@
"name": "Model Name", "name": "Model Name",
"type": "string", "type": "string",
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.", "description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.",
"default": "-hf unsloth/Mistral-Small-3.2-24B-Instruct-2506-GGUF:Q4_K_M -ctx_size 4096", "default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096",
"advanced": false "advanced": false
} }
}, },
+1 -1
View File
@@ -14,7 +14,7 @@
"env": [ "env": [
{ {
"key": "LLAMA_SERVER_CMD_ARGS", "key": "LLAMA_SERVER_CMD_ARGS",
"value": "-hf unsloth/Mistral-Small-3.2-24B-Instruct-2506-GGUF:Q4_K_M -ctx_size 4096" "value": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096"
} }
], ],
"allowedCudaVersions": [ "allowedCudaVersions": [
+1 -1
View File
@@ -33,7 +33,7 @@ WORKDIR /work
ADD ./src /work ADD ./src /work
# Install runpod and its dependencies # Install runpod and its dependencies
RUN pip install -r requirements.txt && chmod +x /work/start.sh RUN pip install -r ./requirements.txt && chmod +x /work/start.sh
# Set the entrypoint # Set the entrypoint
ENTRYPOINT ["/bin/sh", "-c", "/work/start.sh"] ENTRYPOINT ["/bin/sh", "-c", "/work/start.sh"]
+3 -3
View File
@@ -4,8 +4,6 @@
# Serverless llama.cpp inference worker for RunPod # Serverless llama.cpp inference worker for RunPod
[![Runpod badge](https://api.runpod.io/badge/Jacob-ML/inference-worker)](https://console.runpod.io/hub/Jacob-ML/inference-worker)
This repository contains a serverless inference worker for running llama.cpp models on RunPod. It uses the `llama-server` image to provide an API for interacting with the models. This repository contains a serverless inference worker for running llama.cpp models on RunPod. It uses the `llama-server` image to provide an API for interacting with the models.
The following OpenAI API endpoints are supported: The following OpenAI API endpoints are supported:
@@ -24,9 +22,11 @@ Make sure your RunPod worker has access to the network volume, i.e. is located i
The worker can be configured via environment variables set in the RunPod hub configuration: The worker can be configured via environment variables set in the RunPod hub configuration:
- `LLAMA_SERVER_CMD_ARGS`: Command line arguments (argv) for the `llama-server` binary. Example: `-hf /path/to/model.gguf:Q4_K_M -ctx_size 4096`. **IMPORTANT**: Please do not define the port argument here, as the worker will always use port `3098` automatically. - `LLAMA_SERVER_CMD_ARGS`: Command line arguments (argv) for the `llama-server` binary. Example: `-hf /path/to/model.gguf:Q4_K_M --ctx-size 4096`. **IMPORTANT**: Please do not define the port argument here, as the worker will always use port `3098` automatically.
- `MAX_CONCURRENCY`: Maximum number of concurrent requests the worker can handle. Default is `8`. - `MAX_CONCURRENCY`: Maximum number of concurrent requests the worker can handle. Default is `8`.
## License ## License
Please see the [LICENSE](./LICENSE) file for more information. Please see the [LICENSE](./LICENSE) file for more information.
[![Runpod badge](https://api.runpod.io/badge/Jacob-ML/inference-worker)](https://console.runpod.io/hub/Jacob-ML/inference-worker)
-1
View File
@@ -1,4 +1,3 @@
runpod runpod
python-dotenv python-dotenv
openai openai
orjson==3.10.14
+33 -11
View File
@@ -1,24 +1,33 @@
#!/bin/bash #!/bin/bash
# fail on error:
set -e -o pipefail
# This script starts the llama-server with the command line arguments # This script starts the llama-server with the command line arguments
# specified in the environment variable LLAMA_SERVER_CMD_ARGS, ensuring # specified in the environment variable LLAMA_SERVER_CMD_ARGS, ensuring
# that the server listens on port 3098. It also starts the handler.py # that the server listens on port 3098. It also starts the handler.py
# script after the server is up and running. # script after the server is up and running.
cleanup() { cleanup() {
echo "Cleaning up..." echo "start.sh: Cleaning up..."
pkill -P $$ # kill all child processes of the current script pkill -P $$ # kill all child processes of the current script
exit 0 exit 0
} }
# check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS # check if $LLAMA_SERVER_CMD_ARGS is set
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then
echo "Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace." echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096"
LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096"
fi fi
# check if the substring -port is in LLAMA_SERVER_CMD_ARGS # check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"-port"* ]]; then if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then
echo "Error: You must not define -port in LLAMA_SERVER_CMD_ARGS, as port 3098 is required." echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace."
fi
# check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error:
if [[ "$LLAMA_SERVER_CMD_ARGS" == *"--port"* ]]; then
echo "start.sh: Error: You must not define --port in LLAMA_SERVER_CMD_ARGS, as port 3098 is required."
exit 1 exit 1
fi fi
@@ -26,18 +35,27 @@ fi
trap cleanup SIGINT SIGTERM trap cleanup SIGINT SIGTERM
# kill any existing llama-server processes # kill any existing llama-server processes
pgrep llama-server | xargs kill echo "start.sh: Stopping existing llama-server instances (if any)..."
{
pkill llama-server 2>/dev/null
} || {
echo "start.sh: No llama-server running"
}
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS; # we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
# it contains a.e. "-hf modelname -ctx_size 4096". # it contains a.e. "-hf modelname --ctx-size 4096".
echo "start.sh: Running llama-server $LLAMA_SERVER_CMD_ARGS --port 3098"
touch llama.server.log
# We need to pass these arguments to llama-server verbatim. # We need to pass these arguments to llama-server verbatim.
llama-server $LLAMA_SERVER_CMD_ARGS -port 3098 2>&1 | tee llama.server.log & LD_LIBRARY_PATH=/app /app/llama-server $LLAMA_SERVER_CMD_ARGS --port 3098 2>&1 | tee llama.server.log &
LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command
check_server_is_running() { check_server_is_running() {
echo "Checking if llama-server is done initializing..." echo "start.sh: Checking if llama-server is done initializing..."
if cat llama.server.log | grep -q "listening"; then if cat llama.server.log | grep -q "listening"; then
return 0 # success return 0 # success
@@ -46,9 +64,13 @@ check_server_is_running() {
fi fi
} }
echo "start.sh: Waiting for llama-server to start..."
# wait for the server to start # wait for the server to start
while ! check_server_is_running; do while ! check_server_is_running; do
sleep 5 sleep 5
done done
echo "start.sh: llama-server is up and running, delegating to the handler script."
python -u handler.py $1 python -u handler.py $1