7 Commits
7 changed files with 102 additions and 21 deletions
+2 -2
View File
@@ -14,10 +14,10 @@
{ {
"key": "LLAMA_SERVER_CMD_ARGS", "key": "LLAMA_SERVER_CMD_ARGS",
"input": { "input": {
"name": "Model Name", "name": "Command line arguments for llama-server",
"type": "string", "type": "string",
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.", "description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.",
"default": "-hf unsloth/Mistral-Small-3.2-24B-Instruct-2506-GGUF:Q4_K_M -ctx_size 4096", "default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 99",
"advanced": false "advanced": false
} }
}, },
+1 -1
View File
@@ -14,7 +14,7 @@
"env": [ "env": [
{ {
"key": "LLAMA_SERVER_CMD_ARGS", "key": "LLAMA_SERVER_CMD_ARGS",
"value": "-hf unsloth/Mistral-Small-3.2-24B-Instruct-2506-GGUF:Q4_K_M -ctx_size 4096" "value": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096"
} }
], ],
"allowedCudaVersions": [ "allowedCudaVersions": [
+7 -3
View File
@@ -4,8 +4,6 @@
# Serverless llama.cpp inference worker for RunPod # Serverless llama.cpp inference worker for RunPod
[![Runpod badge](https://api.runpod.io/badge/Jacob-ML/inference-worker)](https://console.runpod.io/hub/Jacob-ML/inference-worker)
This repository contains a serverless inference worker for running llama.cpp models on RunPod. It uses the `llama-server` image to provide an API for interacting with the models. This repository contains a serverless inference worker for running llama.cpp models on RunPod. It uses the `llama-server` image to provide an API for interacting with the models.
The following OpenAI API endpoints are supported: The following OpenAI API endpoints are supported:
@@ -15,6 +13,10 @@ The following OpenAI API endpoints are supported:
Streaming responses is also supported. Streaming responses is also supported.
**Important!** This project is still relatively new. Please [open a new issue](https://github.com/Jacob-ML/inference-worker/issues/new) if you encounter any problems in order to get help.
**This is a fork of [SvenBrnn's `runpod-worker-ollama`](https://github.com/SvenBrnn/runpod-worker-ollama).**
## Setup ## Setup
For the setup to work best, it is recommended to use a network volume attached to all workers which stores the model GGUFs and then reference those files in the launch arguments. For the setup to work best, it is recommended to use a network volume attached to all workers which stores the model GGUFs and then reference those files in the launch arguments.
@@ -24,9 +26,11 @@ Make sure your RunPod worker has access to the network volume, i.e. is located i
The worker can be configured via environment variables set in the RunPod hub configuration: The worker can be configured via environment variables set in the RunPod hub configuration:
- `LLAMA_SERVER_CMD_ARGS`: Command line arguments (argv) for the `llama-server` binary. Example: `-hf /path/to/model.gguf:Q4_K_M -ctx_size 4096`. **IMPORTANT**: Please do not define the port argument here, as the worker will always use port `3098` automatically. - `LLAMA_SERVER_CMD_ARGS`: Command line arguments (argv) for the `llama-server` binary. Example: `-hf /path/to/model.gguf:Q4_K_M --ctx-size 4096`. **IMPORTANT**: Please do not define the port argument here, as the worker will always use port `3098` automatically.
- `MAX_CONCURRENCY`: Maximum number of concurrent requests the worker can handle. Default is `8`. - `MAX_CONCURRENCY`: Maximum number of concurrent requests the worker can handle. Default is `8`.
## License ## License
Please see the [LICENSE](./LICENSE) file for more information. Please see the [LICENSE](./LICENSE) file for more information.
[![Runpod badge](https://api.runpod.io/badge/Jacob-ML/inference-worker)](https://console.runpod.io/hub/Jacob-ML/inference-worker)
-1
View File
@@ -21,7 +21,6 @@ Typical usage:
""" """
import json import json
import os
from dotenv import load_dotenv from dotenv import load_dotenv
from openai import OpenAI from openai import OpenAI
+59
View File
@@ -0,0 +1,59 @@
"""
Finds the full LLM GGUF path from the Hugging Face cache.
"""
import os
import argparse
CACHE_DIR = "/runpod-volume/huggingface-cache/hub"
def find_model_path(model_name, gguf_in_repo="model.gguf"):
"""
Find the path to a cached model.
Args:
model_name: The model name from Hugging Face
Returns:
The full path to the cached model, or None if not found
"""
cache_name = model_name.replace("/", "--")
snapshots_dir = os.path.join(
CACHE_DIR, f"models--{cache_name}", "snapshots"
)
if os.path.exists(snapshots_dir):
snapshots = os.listdir(snapshots_dir)
if snapshots:
return os.path.join(snapshots_dir, snapshots[0], gguf_in_repo)
return None
def main():
"""
Main function to find and print the model path.
"""
parser = argparse.ArgumentParser(
description="Find the full GGUF path from the Hugging Face cache."
)
parser.add_argument(
"model", type=str, help="The model name from Hugging Face"
)
parser.add_argument(
"path",
type=str,
help="The path to the GGUF file within the model repository",
)
args = parser.parse_args()
model_path = find_model_path(args.model, args.path)
print(model_path, end="")
if __name__ == "__main__":
main()
-1
View File
@@ -1,4 +1,3 @@
runpod runpod
python-dotenv python-dotenv
openai openai
orjson==3.10.14
+33 -13
View File
@@ -1,7 +1,7 @@
#!/bin/bash #!/bin/bash
# fail on error: # fail on error:
set -e set -e -o pipefail
# This script starts the llama-server with the command line arguments # This script starts the llama-server with the command line arguments
# specified in the environment variable LLAMA_SERVER_CMD_ARGS, ensuring # specified in the environment variable LLAMA_SERVER_CMD_ARGS, ensuring
@@ -9,19 +9,25 @@ set -e
# script after the server is up and running. # script after the server is up and running.
cleanup() { cleanup() {
echo "Cleaning up..." echo "start.sh: Cleaning up..."
pkill -P $$ # kill all child processes of the current script pkill -P $$ # kill all child processes of the current script
exit 0 exit 0
} }
# check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS # check if $LLAMA_SERVER_CMD_ARGS is set
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then
echo "Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace." echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096"
LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 99"
fi fi
# check if the substring -port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error: # check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS
if [[ "$LLAMA_SERVER_CMD_ARGS" == *"-port"* ]]; then if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then
echo "Error: You must not define -port in LLAMA_SERVER_CMD_ARGS, as port 3098 is required." echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace."
fi
# check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error:
if [[ "$LLAMA_SERVER_CMD_ARGS" == *"--port"* ]]; then
echo "start.sh: Error: You must not define --port in LLAMA_SERVER_CMD_ARGS, as port 3098 is required."
exit 1 exit 1
fi fi
@@ -29,18 +35,27 @@ fi
trap cleanup SIGINT SIGTERM trap cleanup SIGINT SIGTERM
# kill any existing llama-server processes # kill any existing llama-server processes
pkill llama-server echo "start.sh: Stopping existing llama-server instances (if any)..."
{
pkill llama-server 2>/dev/null
} || {
echo "start.sh: No llama-server running"
}
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS; # we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
# it contains a.e. "-hf modelname -ctx_size 4096". # it contains a.e. "-hf modelname --ctx-size 4096 -ngl 99".
echo "start.sh: Running llama-server $LLAMA_SERVER_CMD_ARGS --port 3098"
touch llama.server.log
# We need to pass these arguments to llama-server verbatim. # We need to pass these arguments to llama-server verbatim.
/app/llama-server $LLAMA_SERVER_CMD_ARGS -port 3098 2>&1 | tee llama.server.log & LD_LIBRARY_PATH=/app /app/llama-server $LLAMA_SERVER_CMD_ARGS --port 3098 2>&1 | tee llama.server.log &
LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command
check_server_is_running() { check_server_is_running() {
echo "Checking if llama-server is done initializing..." echo "start.sh: Checking if llama-server is done initializing..."
if cat llama.server.log | grep -q "listening"; then if cat llama.server.log | grep -q "listening"; then
return 0 # success return 0 # success
@@ -49,9 +64,14 @@ check_server_is_running() {
fi fi
} }
echo "start.sh: Waiting for llama-server to start..."
# wait for the server to start # wait for the server to start
while ! check_server_is_running; do while ! check_server_is_running; do
sleep 5 # we don't want to lose too much time, so we check very frequently
sleep 0.5
done done
echo "start.sh: llama-server is up and running, delegating to the handler script."
python -u handler.py $1 python -u handler.py $1