2 Commits
4 changed files with 110 additions and 11 deletions
+22 -2
View File
@@ -16,11 +16,31 @@
"input": {
"name": "Command line arguments for llama-server",
"type": "string",
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.",
"default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 99",
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port. If using caching, do not define -hf or -m here.",
"default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 999",
"advanced": false
}
},
{
"key": "LLAMA_CACHED_MODEL",
"input": {
"name": "Hugging Face Hub model name for cached model",
"type": "string",
"description": "Hugging Face Hub model name to use for the cached GGUF model. Leave empty to disable caching. Example: user/model-name",
"default": "",
"advanced": true
}
},
{
"key": "LLAMA_CACHED_GGUF_PATH",
"input": {
"name": "Path to GGUF file in the Hugging Face Hub model repository",
"type": "string",
"description": "Path to the GGUF file in the Hugging Face Hub model repository to use for caching. Example: model.gguf",
"default": "",
"advanced": true
}
},
{
"key": "MAX_CONCURRENCY",
"input": {
+1 -2
View File
@@ -19,8 +19,7 @@ Streaming responses is also supported.
## Setup
For the setup to work best, it is recommended to use a network volume attached to all workers which stores the model GGUFs and then reference those files in the launch arguments.
Make sure your RunPod worker has access to the network volume, i.e. is located in the correct data center.
To get the best performance out of this worker, it is recommended to use cached models. Please see the [cached models documentation](./docs/cached.md) for more information, this is **highly recommended and will save many resources**.
## Configuration
+63
View File
@@ -0,0 +1,63 @@
# Using cached models
## Introduction
The classic way of loading a model from the Hugging Face Hub with the `LLAMA_SERVER_CMD_ARGS` is as follows:
```bash
-hf /path/to/model.gguf:Q4_K_M --ctx-size 4096 # etc...
```
However, this will cause every worker to download the model from the Hugging Face Hub every time it is started, which can be slow and inefficient.
A naive way to cache the model would be to store it on a network volume in RunPod and reference the model files this way:
```bash
-hf /runpod-volume/model.gguf --ctx-size 4096 # etc...
```
Unfortunately, network volume performance is often not sufficient for loading large models, leading to long load times. RunPod introduced a [caching mechanism](https://docs.runpod.io/serverless/endpoints/model-caching) to solve this problem.
The `inference-worker` for llama.cpp now supports this caching mechanism.
## How to use the new caching mechanism
It ships the `src/find_cached.py` script which can be used to reference any Hugging Face model of your choice and get its cached path on the local worker storage.
Here is how the script can be used independently (which you will likely never need to do):
```bash
python3 src/find_cached.py HF_MODEL_ID GGUF_PATH_IN_REPO
```
Example:
```bash
python3 src/find_cached.py unsloth/gemma-3-270m-it-GGUF gemma-3-270m-it-Q8_0.gguf
```
Or, if your model is in a folder (an edge case nobody seems to be thinking about, driving me absolutely crazy):
```bash
python3 src/find_cached.py jacob-ml/jacob-24b models/jacob-24b-q4_k_m.gguf
```
We will now integrate this into our workflow. Hang tight.
## Step-by-step guide
1. First of all, please enter the Hugging Face URL of the model you want to use in RunPod's `Model` field of your worker settings.
Example: For the model `unsloth/gemma-3-270m-it-GGUF`, you would enter `https://huggingface.co/unsloth/gemma-3-270m-it-GGUF`.
2. Now, in the environment variables, do NOT enter the `-hf` argument as before and also do NOT define `-m` in the `LLAMA_SERVER_CMD_ARGS`. The inference worker will take care of that for you.
Instead, set the `LLAMA_CACHED_MODEL` to the model ID, a.e. `unsloth/gemma-3-270m-it-GGUF`. Then, set the `LLAMA_CACHED_GGUF_PATH` to the path of the GGUF file in the repository, e.g. `gemma-3-270m-it-Q8_0.gguf`.
3. Finally, in the `LLAMA_SERVER_CMD_ARGS`, you can now simply add the other arguments you want to use, e.g.:
```bash
--ctx-size 4096 --temp 0.7 --top-p 0.9
```
4. Done! The rest will be handled by the inference worker automatically. When the worker starts, it will resolve the cached model path and launch `llama-server` with the correct arguments.
+24 -7
View File
@@ -14,17 +14,28 @@ cleanup() {
exit 0
}
CACHED_LLAMA_ARGS=""
find_cached_path() {
CACHED_LLAMA_ARGS="-m $(python ./find_cached.py $LLAMA_CACHED_MODEL $LLAMA_CACHED_GGUF_PATH)"
}
# check if $LLAMA_CACHED_MODEL is set and not empty
if [ -n "$LLAMA_CACHED_MODEL" ]; then
echo "start.sh: Caching is enabled. Finding cached model path..."
find_cached_path
echo "start.sh: Using cached model with arguments: $CACHED_LLAMA_ARGS"
else
echo "start.sh: WARNING: Caching is disabled. Please visit the inference-worker README and docs to learn more."
fi
# check if $LLAMA_SERVER_CMD_ARGS is set
if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then
echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
fi
# check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then
echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in the RunPod cache."
fi
# check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error:
if [[ "$LLAMA_SERVER_CMD_ARGS" == *"--port"* ]]; then
echo "start.sh: Error: You must not define --port in LLAMA_SERVER_CMD_ARGS, as port 3098 is required."
@@ -43,14 +54,14 @@ echo "start.sh: Stopping existing llama-server instances (if any)..."
}
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
# it contains a.e. "-hf modelname --ctx-size 4096 -ngl 99".
# it contains a.e. "-hf modelname --ctx-size 4096 -ngl 999".
echo "start.sh: Running llama-server $LLAMA_SERVER_CMD_ARGS --port 3098"
touch llama.server.log
# We need to pass these arguments to llama-server verbatim.
LD_LIBRARY_PATH=/app /app/llama-server $LLAMA_SERVER_CMD_ARGS --port 3098 2>&1 | tee llama.server.log &
LD_LIBRARY_PATH=/app /app/llama-server $CACHED_LLAMA_ARGS $LLAMA_SERVER_CMD_ARGS --port 3098 2>&1 | tee llama.server.log &
LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command
@@ -62,6 +73,12 @@ check_server_is_running() {
else
return 1 # failure
fi
# check if the process is still running
if ! kill -0 $LLAMA_SERVER_PID 2>/dev/null; then
echo "start.sh: Error: llama-server process has exited unexpectedly."
exit 1
fi
}
echo "start.sh: Waiting for llama-server to start..."