diff --git a/.runpod/hub.json b/.runpod/hub.json index bf87c42..3b68b08 100644 --- a/.runpod/hub.json +++ b/.runpod/hub.json @@ -17,7 +17,7 @@ "name": "Model Name", "type": "string", "description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.", - "default": "-hf unsloth/Mistral-Small-3.2-24B-Instruct-2506-GGUF:Q4_K_M -ctx_size 4096", + "default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096", "advanced": false } }, diff --git a/.runpod/tests.json b/.runpod/tests.json index 5fcc642..c60272b 100644 --- a/.runpod/tests.json +++ b/.runpod/tests.json @@ -14,7 +14,7 @@ "env": [ { "key": "LLAMA_SERVER_CMD_ARGS", - "value": "-hf unsloth/Mistral-Small-3.2-24B-Instruct-2506-GGUF:Q4_K_M -ctx_size 4096" + "value": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096" } ], "allowedCudaVersions": [ diff --git a/README.md b/README.md index fb28345..fcf6f71 100644 --- a/README.md +++ b/README.md @@ -4,8 +4,6 @@ # Serverless llama.cpp inference worker for RunPod -[![Runpod badge](https://api.runpod.io/badge/Jacob-ML/inference-worker)](https://console.runpod.io/hub/Jacob-ML/inference-worker) - This repository contains a serverless inference worker for running llama.cpp models on RunPod. It uses the `llama-server` image to provide an API for interacting with the models. The following OpenAI API endpoints are supported: @@ -24,9 +22,11 @@ Make sure your RunPod worker has access to the network volume, i.e. is located i The worker can be configured via environment variables set in the RunPod hub configuration: -- `LLAMA_SERVER_CMD_ARGS`: Command line arguments (argv) for the `llama-server` binary. Example: `-hf /path/to/model.gguf:Q4_K_M -ctx_size 4096`. **IMPORTANT**: Please do not define the port argument here, as the worker will always use port `3098` automatically. +- `LLAMA_SERVER_CMD_ARGS`: Command line arguments (argv) for the `llama-server` binary. Example: `-hf /path/to/model.gguf:Q4_K_M --ctx-size 4096`. **IMPORTANT**: Please do not define the port argument here, as the worker will always use port `3098` automatically. - `MAX_CONCURRENCY`: Maximum number of concurrent requests the worker can handle. Default is `8`. ## License Please see the [LICENSE](./LICENSE) file for more information. + +[![Runpod badge](https://api.runpod.io/badge/Jacob-ML/inference-worker)](https://console.runpod.io/hub/Jacob-ML/inference-worker) diff --git a/src/requirements.txt b/src/requirements.txt index 47eb4ac..ce1c3bf 100644 --- a/src/requirements.txt +++ b/src/requirements.txt @@ -1,4 +1,3 @@ runpod python-dotenv openai -orjson==3.10.14 diff --git a/src/start.sh b/src/start.sh index c06d523..7233d81 100644 --- a/src/start.sh +++ b/src/start.sh @@ -1,7 +1,7 @@ #!/bin/bash # fail on error: -set -e +set -e -o pipefail # This script starts the llama-server with the command line arguments # specified in the environment variable LLAMA_SERVER_CMD_ARGS, ensuring @@ -9,19 +9,19 @@ set -e # script after the server is up and running. cleanup() { - echo "Cleaning up..." + echo "start.sh: Cleaning up..." pkill -P $$ # kill all child processes of the current script exit 0 } # check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then - echo "Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace." + echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace." fi -# check if the substring -port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error: -if [[ "$LLAMA_SERVER_CMD_ARGS" == *"-port"* ]]; then - echo "Error: You must not define -port in LLAMA_SERVER_CMD_ARGS, as port 3098 is required." +# check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error: +if [[ "$LLAMA_SERVER_CMD_ARGS" == *"--port"* ]]; then + echo "start.sh: Error: You must not define --port in LLAMA_SERVER_CMD_ARGS, as port 3098 is required." exit 1 fi @@ -29,18 +29,27 @@ fi trap cleanup SIGINT SIGTERM # kill any existing llama-server processes -pkill llama-server +echo "start.sh: Stopping existing llama-server instances (if any)..." +{ + pkill llama-server 2>/dev/null +} || { + echo "start.sh: No llama-server running" +} # we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS; -# it contains a.e. "-hf modelname -ctx_size 4096". +# it contains a.e. "-hf modelname --ctx-size 4096". + +echo "start.sh: Running llama-server $LLAMA_SERVER_CMD_ARGS --port 3098" + +touch llama.server.log # We need to pass these arguments to llama-server verbatim. -/app/llama-server $LLAMA_SERVER_CMD_ARGS -port 3098 2>&1 | tee llama.server.log & +LD_LIBRARY_PATH=/app /app/llama-server $LLAMA_SERVER_CMD_ARGS --port 3098 2>&1 | tee llama.server.log & LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command check_server_is_running() { - echo "Checking if llama-server is done initializing..." + echo "start.sh: Checking if llama-server is done initializing..." if cat llama.server.log | grep -q "listening"; then return 0 # success @@ -49,9 +58,13 @@ check_server_is_running() { fi } +echo "start.sh: Waiting for llama-server to start..." + # wait for the server to start while ! check_server_is_running; do sleep 5 done +echo "start.sh: llama-server is up and running, delegating to the handler script." + python -u handler.py $1