Compare commits
26
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
029f37afab | ||
|
|
85c10953d6 | ||
|
|
ea82e2a901 | ||
|
|
f890c50c8a | ||
|
|
dd2d52b5e3 | ||
|
|
1196621d77 | ||
|
|
3bcbee2d93 | ||
|
|
b24c32c024 | ||
|
|
c7b115bec0 | ||
|
|
8a2982faa4 | ||
|
|
d4ac091735 | ||
|
|
3d442d9123 | ||
|
|
0c814177fe | ||
|
|
0a59d57025 | ||
|
|
398b59c0a1 | ||
|
|
889d3c5c96 | ||
|
|
41b3b3d02b | ||
|
|
a13d6c1b9f | ||
|
|
403d318ffc | ||
|
|
381ed9ffff | ||
|
|
c42b80ebbd | ||
|
|
40d4097799 | ||
|
|
36a8b32d1e | ||
|
|
5f6c099504 | ||
|
|
be3de61c52 | ||
|
|
558da755c3 |
+25
-5
@@ -1,5 +1,5 @@
|
||||
{
|
||||
"title": "llama.cpp inference",
|
||||
"title": "llama.cpp",
|
||||
"description": "Run llama.cpp inference using serverless RunPod workers!",
|
||||
"type": "serverless",
|
||||
"category": "language",
|
||||
@@ -14,13 +14,33 @@
|
||||
{
|
||||
"key": "LLAMA_SERVER_CMD_ARGS",
|
||||
"input": {
|
||||
"name": "Model Name",
|
||||
"name": "Command line arguments for llama-server",
|
||||
"type": "string",
|
||||
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.",
|
||||
"default": "-hf unsloth/Mistral-Small-3.2-24B-Instruct-2506-GGUF:Q4_K_M -ctx_size 4096",
|
||||
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port. If using caching, do not define -hf or -m here.",
|
||||
"default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 999",
|
||||
"advanced": false
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "LLAMA_CACHED_MODEL",
|
||||
"input": {
|
||||
"name": "Hugging Face Hub model name for cached model",
|
||||
"type": "string",
|
||||
"description": "Hugging Face Hub model name to use for the cached GGUF model. Leave empty to disable caching. Example: user/model-name",
|
||||
"default": "",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "LLAMA_CACHED_GGUF_PATH",
|
||||
"input": {
|
||||
"name": "Path to GGUF file in the Hugging Face Hub model repository",
|
||||
"type": "string",
|
||||
"description": "Path to the GGUF file in the Hugging Face Hub model repository to use for caching. Example: model.gguf",
|
||||
"default": "",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "MAX_CONCURRENCY",
|
||||
"input": {
|
||||
@@ -33,4 +53,4 @@
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+2
-2
@@ -5,7 +5,7 @@
|
||||
"input": {
|
||||
"prompt": "Hi! Who are you?"
|
||||
},
|
||||
"timeout": 120000
|
||||
"timeout": 60000
|
||||
}
|
||||
],
|
||||
"config": {
|
||||
@@ -14,7 +14,7 @@
|
||||
"env": [
|
||||
{
|
||||
"key": "LLAMA_SERVER_CMD_ARGS",
|
||||
"value": "-hf unsloth/Mistral-Small-3.2-24B-Instruct-2506-GGUF:Q4_K_M -ctx_size 4096"
|
||||
"value": "-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
|
||||
}
|
||||
],
|
||||
"allowedCudaVersions": [
|
||||
|
||||
+1
-1
@@ -33,7 +33,7 @@ WORKDIR /work
|
||||
ADD ./src /work
|
||||
|
||||
# Install runpod and its dependencies
|
||||
RUN pip install -r requirements.txt && chmod +x /work/start.sh
|
||||
RUN pip install -r ./requirements.txt && chmod +x /work/start.sh
|
||||
|
||||
# Set the entrypoint
|
||||
ENTRYPOINT ["/bin/sh", "-c", "/work/start.sh"]
|
||||
|
||||
@@ -4,8 +4,6 @@
|
||||
|
||||
# Serverless llama.cpp inference worker for RunPod
|
||||
|
||||
[](https://console.runpod.io/hub/Jacob-ML/inference-worker)
|
||||
|
||||
This repository contains a serverless inference worker for running llama.cpp models on RunPod. It uses the `llama-server` image to provide an API for interacting with the models.
|
||||
The following OpenAI API endpoints are supported:
|
||||
|
||||
@@ -15,18 +13,23 @@ The following OpenAI API endpoints are supported:
|
||||
|
||||
Streaming responses is also supported.
|
||||
|
||||
**Important!** This project is still relatively new. Please [open a new issue](https://github.com/Jacob-ML/inference-worker/issues/new) if you encounter any problems in order to get help.
|
||||
|
||||
**This is a fork of [SvenBrnn's `runpod-worker-ollama`](https://github.com/SvenBrnn/runpod-worker-ollama).**
|
||||
|
||||
## Setup
|
||||
|
||||
For the setup to work best, it is recommended to use a network volume attached to all workers which stores the model GGUFs and then reference those files in the launch arguments.
|
||||
Make sure your RunPod worker has access to the network volume, i.e. is located in the correct data center.
|
||||
To get the best performance out of this worker, it is recommended to use cached models. Please see the [cached models documentation](./docs/cached.md) for more information, this is **highly recommended and will save many resources**.
|
||||
|
||||
## Configuration
|
||||
|
||||
The worker can be configured via environment variables set in the RunPod hub configuration:
|
||||
|
||||
- `LLAMA_SERVER_CMD_ARGS`: Command line arguments (argv) for the `llama-server` binary. Example: `-hf /path/to/model.gguf:Q4_K_M -ctx_size 4096`. **IMPORTANT**: Please do not define the port argument here, as the worker will always use port `3098` automatically.
|
||||
- `LLAMA_SERVER_CMD_ARGS`: Command line arguments (argv) for the `llama-server` binary. Example: `-hf /path/to/model.gguf:Q4_K_M --ctx-size 4096`. **IMPORTANT**: Please do not define the port argument here, as the worker will always use port `3098` automatically.
|
||||
- `MAX_CONCURRENCY`: Maximum number of concurrent requests the worker can handle. Default is `8`.
|
||||
|
||||
## License
|
||||
|
||||
Please see the [LICENSE](./LICENSE) file for more information.
|
||||
|
||||
[](https://console.runpod.io/hub/Jacob-ML/inference-worker)
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
# Using cached models
|
||||
|
||||
## Introduction
|
||||
|
||||
The classic way of loading a model from the Hugging Face Hub with the `LLAMA_SERVER_CMD_ARGS` is as follows:
|
||||
|
||||
```bash
|
||||
-hf /path/to/model.gguf:Q4_K_M --ctx-size 4096 # etc...
|
||||
```
|
||||
|
||||
However, this will cause every worker to download the model from the Hugging Face Hub every time it is started, which can be slow and inefficient.
|
||||
|
||||
A naive way to cache the model would be to store it on a network volume in RunPod and reference the model files this way:
|
||||
|
||||
```bash
|
||||
-m /runpod-volume/model.gguf --ctx-size 4096 # etc...
|
||||
```
|
||||
|
||||
Unfortunately, network volume performance is often not sufficient for loading large models, leading to long load times. RunPod introduced a [caching mechanism](https://docs.runpod.io/serverless/endpoints/model-caching) to solve this problem.
|
||||
|
||||
The `inference-worker` for llama.cpp now supports this caching mechanism.
|
||||
|
||||
## How to use the new caching mechanism
|
||||
|
||||
It ships the `src/find_cached.py` script which can be used to reference any Hugging Face model of your choice and get its cached path on the local worker storage.
|
||||
|
||||
Here is how the script can be used independently (which you will likely never need to do):
|
||||
|
||||
```bash
|
||||
python3 src/find_cached.py HF_MODEL_ID GGUF_PATH_IN_REPO
|
||||
```
|
||||
|
||||
Example:
|
||||
|
||||
```bash
|
||||
python3 src/find_cached.py unsloth/gemma-3-270m-it-GGUF gemma-3-270m-it-Q8_0.gguf
|
||||
```
|
||||
|
||||
Or, if your model is in a folder (an edge case nobody seems to be thinking about, driving me absolutely crazy):
|
||||
|
||||
```bash
|
||||
python3 src/find_cached.py jacob-ml/jacob-24b models/jacob-24b-q4_k_m.gguf
|
||||
```
|
||||
|
||||
We will now integrate this into our workflow. Hang tight.
|
||||
|
||||
## Step-by-step guide
|
||||
|
||||
1. First of all, please enter the Hugging Face URL of the model you want to use in RunPod's `Model` field of your worker settings.
|
||||
|
||||
Example: For the model `unsloth/gemma-3-270m-it-GGUF`, you would enter `https://huggingface.co/unsloth/gemma-3-270m-it-GGUF`.
|
||||
|
||||
2. Now, in the environment variables, do NOT enter the `-hf` argument as before and also do NOT define `-m` in the `LLAMA_SERVER_CMD_ARGS`. The inference worker will take care of that for you.
|
||||
|
||||
Instead, set the `LLAMA_CACHED_MODEL` to the model ID, a.e. `unsloth/gemma-3-270m-it-GGUF`. Then, set the `LLAMA_CACHED_GGUF_PATH` to the path of the GGUF file in the repository, e.g. `gemma-3-270m-it-Q8_0.gguf`.
|
||||
|
||||
3. Finally, in the `LLAMA_SERVER_CMD_ARGS`, you can now simply add the other arguments you want to use, e.g.:
|
||||
|
||||
```bash
|
||||
--ctx-size 4096 --temp 0.7 --top-p 0.9
|
||||
```
|
||||
|
||||
4. Done! The rest will be handled by the inference worker automatically. When the worker starts, it will resolve the cached model path and launch `llama-server` with the correct arguments.
|
||||
@@ -0,0 +1,6 @@
|
||||
"""
|
||||
This is an empty file used to mark the repository as a Runpod-compatible
|
||||
serverless endpoint because they won't stop pretending it's not.
|
||||
|
||||
I'm tired.
|
||||
"""
|
||||
+1
-2
@@ -21,7 +21,6 @@ Typical usage:
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from openai import OpenAI
|
||||
@@ -29,7 +28,7 @@ from utils import JobInput
|
||||
|
||||
client = OpenAI(
|
||||
base_url="http://localhost:3098/v1/",
|
||||
api_key="",
|
||||
api_key="none",
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,66 @@
|
||||
"""
|
||||
Finds the full LLM GGUF path from the Hugging Face cache.
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import argparse
|
||||
|
||||
CACHE_DIR = "/runpod-volume/huggingface-cache/hub"
|
||||
|
||||
|
||||
def find_model_path(model_name, gguf_in_repo="model.gguf"):
|
||||
"""
|
||||
Find the path to a cached model.
|
||||
|
||||
Args:
|
||||
model_name: The model name from Hugging Face
|
||||
|
||||
Returns:
|
||||
The full path to the cached model, or None if not found
|
||||
"""
|
||||
|
||||
cache_name = model_name.replace("/", "--").lower()
|
||||
snapshots_dir = os.path.join(
|
||||
CACHE_DIR, f"models--{cache_name}", "snapshots"
|
||||
)
|
||||
|
||||
if os.path.exists(snapshots_dir):
|
||||
snapshots = os.listdir(snapshots_dir)
|
||||
|
||||
if snapshots:
|
||||
return os.path.join(snapshots_dir, snapshots[0], gguf_in_repo)
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def main():
|
||||
"""
|
||||
Main function to find and print the model path.
|
||||
"""
|
||||
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Find the full GGUF path from the Hugging Face cache."
|
||||
)
|
||||
parser.add_argument(
|
||||
"model", type=str, help="The model name from Hugging Face"
|
||||
)
|
||||
parser.add_argument(
|
||||
"path",
|
||||
type=str,
|
||||
help="The path to the GGUF file within the model repository",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
model_path = find_model_path(args.model, args.path)
|
||||
if model_path is None:
|
||||
print(
|
||||
f"Error: Cached model not found. Model='{args.model}', GGUF='{args.path}', Cache dir='{CACHE_DIR}'",
|
||||
file=sys.stderr,
|
||||
)
|
||||
sys.exit(1)
|
||||
print(model_path, end="")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,4 +1,3 @@
|
||||
runpod
|
||||
python-dotenv
|
||||
openai
|
||||
orjson==3.10.14
|
||||
runpod==1.9.1
|
||||
python-dotenv==1.2.2
|
||||
openai==2.41.1
|
||||
|
||||
+67
-12
@@ -1,24 +1,50 @@
|
||||
#!/bin/bash
|
||||
|
||||
# fail on error:
|
||||
set -e -o pipefail
|
||||
|
||||
# This script starts the llama-server with the command line arguments
|
||||
# specified in the environment variable LLAMA_SERVER_CMD_ARGS, ensuring
|
||||
# that the server listens on port 3098. It also starts the handler.py
|
||||
# script after the server is up and running.
|
||||
|
||||
cleanup() {
|
||||
echo "Cleaning up..."
|
||||
echo "start.sh: Cleaning up..."
|
||||
pkill -P $$ # kill all child processes of the current script
|
||||
exit 0
|
||||
}
|
||||
|
||||
# check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS
|
||||
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then
|
||||
echo "Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace."
|
||||
CACHED_LLAMA_ARGS=""
|
||||
|
||||
find_cached_path() {
|
||||
local model_path
|
||||
model_path=$(python ./find_cached.py "$LLAMA_CACHED_MODEL" "$LLAMA_CACHED_GGUF_PATH")
|
||||
if [ $? -ne 0 ] || [ -z "$model_path" ]; then
|
||||
echo "start.sh: Error: Could not resolve cached model path. Check that LLAMA_CACHED_MODEL and LLAMA_CACHED_GGUF_PATH are correct and the model is fully cached on the network volume."
|
||||
exit 1
|
||||
fi
|
||||
CACHED_LLAMA_ARGS="-m $model_path"
|
||||
}
|
||||
|
||||
# check if $LLAMA_CACHED_MODEL is set and not empty
|
||||
if [ -n "$LLAMA_CACHED_MODEL" ]; then
|
||||
echo "start.sh: Caching is enabled. Finding cached model path..."
|
||||
find_cached_path
|
||||
|
||||
echo "start.sh: Using cached model with arguments: $CACHED_LLAMA_ARGS"
|
||||
else
|
||||
echo "start.sh: WARNING: Caching is disabled. Please visit the inference-worker README and docs to learn more."
|
||||
fi
|
||||
|
||||
# check if the substring -port is in LLAMA_SERVER_CMD_ARGS
|
||||
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"-port"* ]]; then
|
||||
echo "Error: You must not define -port in LLAMA_SERVER_CMD_ARGS, as port 3098 is required."
|
||||
# check if $LLAMA_SERVER_CMD_ARGS is set
|
||||
if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then
|
||||
echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
|
||||
LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
|
||||
fi
|
||||
|
||||
# check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error:
|
||||
if [[ "$LLAMA_SERVER_CMD_ARGS" == *"--port"* ]]; then
|
||||
echo "start.sh: Error: You must not define --port in LLAMA_SERVER_CMD_ARGS, as port 3098 is required."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
@@ -26,29 +52,58 @@ fi
|
||||
trap cleanup SIGINT SIGTERM
|
||||
|
||||
# kill any existing llama-server processes
|
||||
pgrep llama-server | xargs kill
|
||||
echo "start.sh: Stopping existing llama-server instances (if any)..."
|
||||
{
|
||||
pkill llama-server 2>/dev/null
|
||||
} || {
|
||||
echo "start.sh: No llama-server running"
|
||||
}
|
||||
|
||||
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
|
||||
# it contains a.e. "-hf modelname -ctx_size 4096".
|
||||
# it contains a.e. "-hf modelname --ctx-size 4096 -ngl 999".
|
||||
|
||||
echo "start.sh: Running /app/llama-server $CACHED_LLAMA_ARGS $LLAMA_SERVER_CMD_ARGS --port 3098"
|
||||
|
||||
touch llama.server.log
|
||||
|
||||
# We need to pass these arguments to llama-server verbatim.
|
||||
llama-server $LLAMA_SERVER_CMD_ARGS -port 3098 2>&1 | tee llama.server.log &
|
||||
LD_LIBRARY_PATH=/app /app/llama-server $CACHED_LLAMA_ARGS $LLAMA_SERVER_CMD_ARGS --port 3098 2>&1 | tee llama.server.log &
|
||||
|
||||
LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command
|
||||
|
||||
tries_so_far=0
|
||||
|
||||
check_server_is_running() {
|
||||
echo "Checking if llama-server is done initializing..."
|
||||
echo "start.sh: Checking if llama-server is done initializing..."
|
||||
|
||||
if cat llama.server.log | grep -q "listening"; then
|
||||
return 0 # success
|
||||
else
|
||||
return 1 # failure
|
||||
fi
|
||||
|
||||
tries_so_far=$((tries_so_far + 1))
|
||||
|
||||
if [ $tries_so_far -ge 120 ]; then
|
||||
echo "start.sh: Error: llama-server did not start within 60 seconds."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# check if the process is still running
|
||||
if ! kill -0 $LLAMA_SERVER_PID 2>/dev/null; then
|
||||
echo "start.sh: Error: llama-server process has exited unexpectedly."
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
echo "start.sh: Waiting for llama-server to start..."
|
||||
|
||||
# wait for the server to start
|
||||
while ! check_server_is_running; do
|
||||
sleep 5
|
||||
# we don't want to lose too much time, so we check very frequently
|
||||
sleep 0.5
|
||||
done
|
||||
|
||||
echo "start.sh: llama-server is up and running, delegating to the handler script."
|
||||
|
||||
python -u handler.py $1
|
||||
|
||||
Reference in New Issue
Block a user