23 Commits
Author SHA1 Message Date
MagnusandGitHub 029f37afab pin dependencies 2026-06-14 18:03:54 +02:00
MagnusandGitHub 85c10953d6 add pseudo API key so the client doesn't raise an error
this seems to have been introduced in a recent OpenAI SDK update
2026-06-14 18:00:42 +02:00
MagnusandGitHub ea82e2a901 remove "inference" from title to make it more consistent with the other runpod repos on the hub 2026-06-03 18:22:06 +02:00
MagnusandGitHub f890c50c8a Merge pull request #3 from cardinalfan1/claude/fix-runpod-startup-xxJqT
Fix -m None passed to llama-server when cached model not found
2026-06-03 18:12:26 +02:00
Claude dd2d52b5e3 Fix -m None passed to llama-server when cached model not found
When LLAMA_CACHED_MODEL is set but the model isn't present in the cache,
find_cached.py was printing Python's None as the string "None", causing
start.sh to pass "-m None" to llama-server.

- find_cached.py: print an error to stderr and exit 1 when the model path
  cannot be resolved, instead of printing "None"
- start.sh: capture find_cached.py output into a local variable, check
  the exit code and guard against empty output before constructing
  CACHED_LLAMA_ARGS; also quote the env-var expansions to handle spaces

https://claude.ai/code/session_011ny5CFYnrzPbRneSzio5CR
2026-04-15 19:34:58 +00:00
mags0ftandGitHub 1196621d77 Merge pull request #1 from aparmar2000/patch-1
automatically convert path to lowercase
2026-03-10 19:11:04 +01:00
Ashan ParmarandGitHub 3bcbee2d93 Update find_cached.py
The cache name is lowercase
2026-02-20 18:04:59 -07:00
mags0ft b24c32c024 re-activate automatic testing 2026-02-12 08:45:54 +01:00
mags0ftandGitHub c7b115bec0 fix typo in cached models docs 2026-02-11 02:14:18 +01:00
mags0ft 8a2982faa4 adjust incorrect llama-server start command log 2025-12-27 23:51:00 +01:00
mags0ft d4ac091735 add timeout for llama-server startup to prevent indefinite waiting 2025-12-27 18:00:58 +01:00
mags0ft 3d442d9123 add error handling for unexpected llama-server process exit 2025-12-26 14:51:06 +01:00
mags0ft 0c814177fe add caching support for models in start.sh and update documentation 2025-12-26 14:49:03 +01:00
mags0ft 0a59d57025 rename tests.json temporarily to skip apparently buggy RunPod CI
The RunPod support told me to do this until the issues are resolved.
2025-12-26 11:37:23 +01:00
mags0ft 398b59c0a1 update test timeout and default LLAMA_SERVER_CMD_ARGS for improved performance 2025-12-25 19:17:31 +01:00
mags0ft 889d3c5c96 move find_cached.py to src 2025-12-18 10:57:10 +01:00
mags0ft 41b3b3d02b add helper script to use cached models 2025-12-18 10:50:33 +01:00
mags0ftandGitHub a13d6c1b9f add issue note and fork explanation to README 2025-11-24 22:24:46 +01:00
mags0ft 403d318ffc update default args to include -ngl 99, improve startup sleep duration 2025-11-19 19:16:18 +01:00
mags0ft 381ed9ffff prepare for production release 2025-11-19 18:40:00 +01:00
mags0ft c42b80ebbd add default behavior for undefined LLAMA_SERVER_CMD_ARGS 2025-11-19 18:11:36 +01:00
mags0ft 40d4097799 fix several major bugs in startup script 2025-11-19 17:47:06 +01:00
mags0ft 36a8b32d1e fix: ensure script fails on error by setting 'set -e' 2025-11-15 16:07:54 +01:00
8 changed files with 235 additions and 30 deletions
+24 -4
View File
@@ -1,5 +1,5 @@
{ {
"title": "llama.cpp inference", "title": "llama.cpp",
"description": "Run llama.cpp inference using serverless RunPod workers!", "description": "Run llama.cpp inference using serverless RunPod workers!",
"type": "serverless", "type": "serverless",
"category": "language", "category": "language",
@@ -14,13 +14,33 @@
{ {
"key": "LLAMA_SERVER_CMD_ARGS", "key": "LLAMA_SERVER_CMD_ARGS",
"input": { "input": {
"name": "Model Name", "name": "Command line arguments for llama-server",
"type": "string", "type": "string",
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.", "description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port. If using caching, do not define -hf or -m here.",
"default": "-hf unsloth/Mistral-Small-3.2-24B-Instruct-2506-GGUF:Q4_K_M -ctx_size 4096", "default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 999",
"advanced": false "advanced": false
} }
}, },
{
"key": "LLAMA_CACHED_MODEL",
"input": {
"name": "Hugging Face Hub model name for cached model",
"type": "string",
"description": "Hugging Face Hub model name to use for the cached GGUF model. Leave empty to disable caching. Example: user/model-name",
"default": "",
"advanced": true
}
},
{
"key": "LLAMA_CACHED_GGUF_PATH",
"input": {
"name": "Path to GGUF file in the Hugging Face Hub model repository",
"type": "string",
"description": "Path to the GGUF file in the Hugging Face Hub model repository to use for caching. Example: model.gguf",
"default": "",
"advanced": true
}
},
{ {
"key": "MAX_CONCURRENCY", "key": "MAX_CONCURRENCY",
"input": { "input": {
+2 -2
View File
@@ -5,7 +5,7 @@
"input": { "input": {
"prompt": "Hi! Who are you?" "prompt": "Hi! Who are you?"
}, },
"timeout": 120000 "timeout": 60000
} }
], ],
"config": { "config": {
@@ -14,7 +14,7 @@
"env": [ "env": [
{ {
"key": "LLAMA_SERVER_CMD_ARGS", "key": "LLAMA_SERVER_CMD_ARGS",
"value": "-hf unsloth/Mistral-Small-3.2-24B-Instruct-2506-GGUF:Q4_K_M -ctx_size 4096" "value": "-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
} }
], ],
"allowedCudaVersions": [ "allowedCudaVersions": [
+8 -5
View File
@@ -4,8 +4,6 @@
# Serverless llama.cpp inference worker for RunPod # Serverless llama.cpp inference worker for RunPod
[![Runpod badge](https://api.runpod.io/badge/Jacob-ML/inference-worker)](https://console.runpod.io/hub/Jacob-ML/inference-worker)
This repository contains a serverless inference worker for running llama.cpp models on RunPod. It uses the `llama-server` image to provide an API for interacting with the models. This repository contains a serverless inference worker for running llama.cpp models on RunPod. It uses the `llama-server` image to provide an API for interacting with the models.
The following OpenAI API endpoints are supported: The following OpenAI API endpoints are supported:
@@ -15,18 +13,23 @@ The following OpenAI API endpoints are supported:
Streaming responses is also supported. Streaming responses is also supported.
**Important!** This project is still relatively new. Please [open a new issue](https://github.com/Jacob-ML/inference-worker/issues/new) if you encounter any problems in order to get help.
**This is a fork of [SvenBrnn's `runpod-worker-ollama`](https://github.com/SvenBrnn/runpod-worker-ollama).**
## Setup ## Setup
For the setup to work best, it is recommended to use a network volume attached to all workers which stores the model GGUFs and then reference those files in the launch arguments. To get the best performance out of this worker, it is recommended to use cached models. Please see the [cached models documentation](./docs/cached.md) for more information, this is **highly recommended and will save many resources**.
Make sure your RunPod worker has access to the network volume, i.e. is located in the correct data center.
## Configuration ## Configuration
The worker can be configured via environment variables set in the RunPod hub configuration: The worker can be configured via environment variables set in the RunPod hub configuration:
- `LLAMA_SERVER_CMD_ARGS`: Command line arguments (argv) for the `llama-server` binary. Example: `-hf /path/to/model.gguf:Q4_K_M -ctx_size 4096`. **IMPORTANT**: Please do not define the port argument here, as the worker will always use port `3098` automatically. - `LLAMA_SERVER_CMD_ARGS`: Command line arguments (argv) for the `llama-server` binary. Example: `-hf /path/to/model.gguf:Q4_K_M --ctx-size 4096`. **IMPORTANT**: Please do not define the port argument here, as the worker will always use port `3098` automatically.
- `MAX_CONCURRENCY`: Maximum number of concurrent requests the worker can handle. Default is `8`. - `MAX_CONCURRENCY`: Maximum number of concurrent requests the worker can handle. Default is `8`.
## License ## License
Please see the [LICENSE](./LICENSE) file for more information. Please see the [LICENSE](./LICENSE) file for more information.
[![Runpod badge](https://api.runpod.io/badge/Jacob-ML/inference-worker)](https://console.runpod.io/hub/Jacob-ML/inference-worker)
+63
View File
@@ -0,0 +1,63 @@
# Using cached models
## Introduction
The classic way of loading a model from the Hugging Face Hub with the `LLAMA_SERVER_CMD_ARGS` is as follows:
```bash
-hf /path/to/model.gguf:Q4_K_M --ctx-size 4096 # etc...
```
However, this will cause every worker to download the model from the Hugging Face Hub every time it is started, which can be slow and inefficient.
A naive way to cache the model would be to store it on a network volume in RunPod and reference the model files this way:
```bash
-m /runpod-volume/model.gguf --ctx-size 4096 # etc...
```
Unfortunately, network volume performance is often not sufficient for loading large models, leading to long load times. RunPod introduced a [caching mechanism](https://docs.runpod.io/serverless/endpoints/model-caching) to solve this problem.
The `inference-worker` for llama.cpp now supports this caching mechanism.
## How to use the new caching mechanism
It ships the `src/find_cached.py` script which can be used to reference any Hugging Face model of your choice and get its cached path on the local worker storage.
Here is how the script can be used independently (which you will likely never need to do):
```bash
python3 src/find_cached.py HF_MODEL_ID GGUF_PATH_IN_REPO
```
Example:
```bash
python3 src/find_cached.py unsloth/gemma-3-270m-it-GGUF gemma-3-270m-it-Q8_0.gguf
```
Or, if your model is in a folder (an edge case nobody seems to be thinking about, driving me absolutely crazy):
```bash
python3 src/find_cached.py jacob-ml/jacob-24b models/jacob-24b-q4_k_m.gguf
```
We will now integrate this into our workflow. Hang tight.
## Step-by-step guide
1. First of all, please enter the Hugging Face URL of the model you want to use in RunPod's `Model` field of your worker settings.
Example: For the model `unsloth/gemma-3-270m-it-GGUF`, you would enter `https://huggingface.co/unsloth/gemma-3-270m-it-GGUF`.
2. Now, in the environment variables, do NOT enter the `-hf` argument as before and also do NOT define `-m` in the `LLAMA_SERVER_CMD_ARGS`. The inference worker will take care of that for you.
Instead, set the `LLAMA_CACHED_MODEL` to the model ID, a.e. `unsloth/gemma-3-270m-it-GGUF`. Then, set the `LLAMA_CACHED_GGUF_PATH` to the path of the GGUF file in the repository, e.g. `gemma-3-270m-it-Q8_0.gguf`.
3. Finally, in the `LLAMA_SERVER_CMD_ARGS`, you can now simply add the other arguments you want to use, e.g.:
```bash
--ctx-size 4096 --temp 0.7 --top-p 0.9
```
4. Done! The rest will be handled by the inference worker automatically. When the worker starts, it will resolve the cached model path and launch `llama-server` with the correct arguments.
+1 -2
View File
@@ -21,7 +21,6 @@ Typical usage:
""" """
import json import json
import os
from dotenv import load_dotenv from dotenv import load_dotenv
from openai import OpenAI from openai import OpenAI
@@ -29,7 +28,7 @@ from utils import JobInput
client = OpenAI( client = OpenAI(
base_url="http://localhost:3098/v1/", base_url="http://localhost:3098/v1/",
api_key="", api_key="none",
) )
+66
View File
@@ -0,0 +1,66 @@
"""
Finds the full LLM GGUF path from the Hugging Face cache.
"""
import os
import sys
import argparse
CACHE_DIR = "/runpod-volume/huggingface-cache/hub"
def find_model_path(model_name, gguf_in_repo="model.gguf"):
"""
Find the path to a cached model.
Args:
model_name: The model name from Hugging Face
Returns:
The full path to the cached model, or None if not found
"""
cache_name = model_name.replace("/", "--").lower()
snapshots_dir = os.path.join(
CACHE_DIR, f"models--{cache_name}", "snapshots"
)
if os.path.exists(snapshots_dir):
snapshots = os.listdir(snapshots_dir)
if snapshots:
return os.path.join(snapshots_dir, snapshots[0], gguf_in_repo)
return None
def main():
"""
Main function to find and print the model path.
"""
parser = argparse.ArgumentParser(
description="Find the full GGUF path from the Hugging Face cache."
)
parser.add_argument(
"model", type=str, help="The model name from Hugging Face"
)
parser.add_argument(
"path",
type=str,
help="The path to the GGUF file within the model repository",
)
args = parser.parse_args()
model_path = find_model_path(args.model, args.path)
if model_path is None:
print(
f"Error: Cached model not found. Model='{args.model}', GGUF='{args.path}', Cache dir='{CACHE_DIR}'",
file=sys.stderr,
)
sys.exit(1)
print(model_path, end="")
if __name__ == "__main__":
main()
+3 -4
View File
@@ -1,4 +1,3 @@
runpod runpod==1.9.1
python-dotenv python-dotenv==1.2.2
openai openai==2.41.1
orjson==3.10.14
+67 -12
View File
@@ -1,24 +1,50 @@
#!/bin/bash #!/bin/bash
# fail on error:
set -e -o pipefail
# This script starts the llama-server with the command line arguments # This script starts the llama-server with the command line arguments
# specified in the environment variable LLAMA_SERVER_CMD_ARGS, ensuring # specified in the environment variable LLAMA_SERVER_CMD_ARGS, ensuring
# that the server listens on port 3098. It also starts the handler.py # that the server listens on port 3098. It also starts the handler.py
# script after the server is up and running. # script after the server is up and running.
cleanup() { cleanup() {
echo "Cleaning up..." echo "start.sh: Cleaning up..."
pkill -P $$ # kill all child processes of the current script pkill -P $$ # kill all child processes of the current script
exit 0 exit 0
} }
# check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS CACHED_LLAMA_ARGS=""
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then
echo "Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace." find_cached_path() {
local model_path
model_path=$(python ./find_cached.py "$LLAMA_CACHED_MODEL" "$LLAMA_CACHED_GGUF_PATH")
if [ $? -ne 0 ] || [ -z "$model_path" ]; then
echo "start.sh: Error: Could not resolve cached model path. Check that LLAMA_CACHED_MODEL and LLAMA_CACHED_GGUF_PATH are correct and the model is fully cached on the network volume."
exit 1
fi
CACHED_LLAMA_ARGS="-m $model_path"
}
# check if $LLAMA_CACHED_MODEL is set and not empty
if [ -n "$LLAMA_CACHED_MODEL" ]; then
echo "start.sh: Caching is enabled. Finding cached model path..."
find_cached_path
echo "start.sh: Using cached model with arguments: $CACHED_LLAMA_ARGS"
else
echo "start.sh: WARNING: Caching is disabled. Please visit the inference-worker README and docs to learn more."
fi fi
# check if the substring -port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error: # check if $LLAMA_SERVER_CMD_ARGS is set
if [[ "$LLAMA_SERVER_CMD_ARGS" == *"-port"* ]]; then if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then
echo "Error: You must not define -port in LLAMA_SERVER_CMD_ARGS, as port 3098 is required." echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
fi
# check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error:
if [[ "$LLAMA_SERVER_CMD_ARGS" == *"--port"* ]]; then
echo "start.sh: Error: You must not define --port in LLAMA_SERVER_CMD_ARGS, as port 3098 is required."
exit 1 exit 1
fi fi
@@ -26,29 +52,58 @@ fi
trap cleanup SIGINT SIGTERM trap cleanup SIGINT SIGTERM
# kill any existing llama-server processes # kill any existing llama-server processes
pkill llama-server echo "start.sh: Stopping existing llama-server instances (if any)..."
{
pkill llama-server 2>/dev/null
} || {
echo "start.sh: No llama-server running"
}
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS; # we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
# it contains a.e. "-hf modelname -ctx_size 4096". # it contains a.e. "-hf modelname --ctx-size 4096 -ngl 999".
echo "start.sh: Running /app/llama-server $CACHED_LLAMA_ARGS $LLAMA_SERVER_CMD_ARGS --port 3098"
touch llama.server.log
# We need to pass these arguments to llama-server verbatim. # We need to pass these arguments to llama-server verbatim.
/app/llama-server $LLAMA_SERVER_CMD_ARGS -port 3098 2>&1 | tee llama.server.log & LD_LIBRARY_PATH=/app /app/llama-server $CACHED_LLAMA_ARGS $LLAMA_SERVER_CMD_ARGS --port 3098 2>&1 | tee llama.server.log &
LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command
tries_so_far=0
check_server_is_running() { check_server_is_running() {
echo "Checking if llama-server is done initializing..." echo "start.sh: Checking if llama-server is done initializing..."
if cat llama.server.log | grep -q "listening"; then if cat llama.server.log | grep -q "listening"; then
return 0 # success return 0 # success
else else
return 1 # failure return 1 # failure
fi fi
tries_so_far=$((tries_so_far + 1))
if [ $tries_so_far -ge 120 ]; then
echo "start.sh: Error: llama-server did not start within 60 seconds."
exit 1
fi
# check if the process is still running
if ! kill -0 $LLAMA_SERVER_PID 2>/dev/null; then
echo "start.sh: Error: llama-server process has exited unexpectedly."
exit 1
fi
} }
echo "start.sh: Waiting for llama-server to start..."
# wait for the server to start # wait for the server to start
while ! check_server_is_running; do while ! check_server_is_running; do
sleep 5 # we don't want to lose too much time, so we check very frequently
sleep 0.5
done done
echo "start.sh: llama-server is up and running, delegating to the handler script."
python -u handler.py $1 python -u handler.py $1