From 0c814177fed22c2c1f43c788dc3cc2a0a2cfbddb Mon Sep 17 00:00:00 2001 From: mags0ft Date: Fri, 26 Dec 2025 14:49:03 +0100 Subject: [PATCH] add caching support for models in start.sh and update documentation --- .runpod/hub.json | 24 ++++++++++++++++-- README.md | 3 +-- docs/cached.md | 63 ++++++++++++++++++++++++++++++++++++++++++++++++ src/start.sh | 25 +++++++++++++------ 4 files changed, 104 insertions(+), 11 deletions(-) create mode 100644 docs/cached.md diff --git a/.runpod/hub.json b/.runpod/hub.json index f66d397..5c77919 100644 --- a/.runpod/hub.json +++ b/.runpod/hub.json @@ -16,11 +16,31 @@ "input": { "name": "Command line arguments for llama-server", "type": "string", - "description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.", - "default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 99", + "description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port. If using caching, do not define -hf or -m here.", + "default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 999", "advanced": false } }, + { + "key": "LLAMA_CACHED_MODEL", + "input": { + "name": "Hugging Face Hub model name for cached model", + "type": "string", + "description": "Hugging Face Hub model name to use for the cached GGUF model. Leave empty to disable caching. Example: user/model-name", + "default": "", + "advanced": true + } + }, + { + "key": "LLAMA_CACHED_GGUF_PATH", + "input": { + "name": "Path to GGUF file in the Hugging Face Hub model repository", + "type": "string", + "description": "Path to the GGUF file in the Hugging Face Hub model repository to use for caching. Example: model.gguf", + "default": "", + "advanced": true + } + }, { "key": "MAX_CONCURRENCY", "input": { diff --git a/README.md b/README.md index 79c288e..8b3ee30 100644 --- a/README.md +++ b/README.md @@ -19,8 +19,7 @@ Streaming responses is also supported. ## Setup -For the setup to work best, it is recommended to use a network volume attached to all workers which stores the model GGUFs and then reference those files in the launch arguments. -Make sure your RunPod worker has access to the network volume, i.e. is located in the correct data center. +To get the best performance out of this worker, it is recommended to use cached models. Please see the [cached models documentation](./docs/cached.md) for more information, this is **highly recommended and will save many resources**. ## Configuration diff --git a/docs/cached.md b/docs/cached.md new file mode 100644 index 0000000..fcd5ae2 --- /dev/null +++ b/docs/cached.md @@ -0,0 +1,63 @@ +# Using cached models + +## Introduction + +The classic way of loading a model from the Hugging Face Hub with the `LLAMA_SERVER_CMD_ARGS` is as follows: + +```bash +-hf /path/to/model.gguf:Q4_K_M --ctx-size 4096 # etc... +``` + +However, this will cause every worker to download the model from the Hugging Face Hub every time it is started, which can be slow and inefficient. + +A naive way to cache the model would be to store it on a network volume in RunPod and reference the model files this way: + +```bash +-hf /runpod-volume/model.gguf --ctx-size 4096 # etc... +``` + +Unfortunately, network volume performance is often not sufficient for loading large models, leading to long load times. RunPod introduced a [caching mechanism](https://docs.runpod.io/serverless/endpoints/model-caching) to solve this problem. + +The `inference-worker` for llama.cpp now supports this caching mechanism. + +## How to use the new caching mechanism + +It ships the `src/find_cached.py` script which can be used to reference any Hugging Face model of your choice and get its cached path on the local worker storage. + +Here is how the script can be used independently (which you will likely never need to do): + +```bash +python3 src/find_cached.py HF_MODEL_ID GGUF_PATH_IN_REPO +``` + +Example: + +```bash +python3 src/find_cached.py unsloth/gemma-3-270m-it-GGUF gemma-3-270m-it-Q8_0.gguf +``` + +Or, if your model is in a folder (an edge case nobody seems to be thinking about, driving me absolutely crazy): + +```bash +python3 src/find_cached.py jacob-ml/jacob-24b models/jacob-24b-q4_k_m.gguf +``` + +We will now integrate this into our workflow. Hang tight. + +## Step-by-step guide + +1. First of all, please enter the Hugging Face URL of the model you want to use in RunPod's `Model` field of your worker settings. + + Example: For the model `unsloth/gemma-3-270m-it-GGUF`, you would enter `https://huggingface.co/unsloth/gemma-3-270m-it-GGUF`. + +2. Now, in the environment variables, do NOT enter the `-hf` argument as before and also do NOT define `-m` in the `LLAMA_SERVER_CMD_ARGS`. The inference worker will take care of that for you. + + Instead, set the `LLAMA_CACHED_MODEL` to the model ID, a.e. `unsloth/gemma-3-270m-it-GGUF`. Then, set the `LLAMA_CACHED_GGUF_PATH` to the path of the GGUF file in the repository, e.g. `gemma-3-270m-it-Q8_0.gguf`. + +3. Finally, in the `LLAMA_SERVER_CMD_ARGS`, you can now simply add the other arguments you want to use, e.g.: + + ```bash + --ctx-size 4096 --temp 0.7 --top-p 0.9 + ``` + +4. Done! The rest will be handled by the inference worker automatically. When the worker starts, it will resolve the cached model path and launch `llama-server` with the correct arguments. diff --git a/src/start.sh b/src/start.sh index 285ee0d..488e7c1 100644 --- a/src/start.sh +++ b/src/start.sh @@ -14,17 +14,28 @@ cleanup() { exit 0 } +CACHED_LLAMA_ARGS="" + +find_cached_path() { + CACHED_LLAMA_ARGS="-m $(python ./find_cached.py $LLAMA_CACHED_MODEL $LLAMA_CACHED_GGUF_PATH)" +} + +# check if $LLAMA_CACHED_MODEL is set and not empty +if [ -n "$LLAMA_CACHED_MODEL" ]; then + echo "start.sh: Caching is enabled. Finding cached model path..." + find_cached_path + + echo "start.sh: Using cached model with arguments: $CACHED_LLAMA_ARGS" +else + echo "start.sh: WARNING: Caching is disabled. Please visit the inference-worker README and docs to learn more." +fi + # check if $LLAMA_SERVER_CMD_ARGS is set if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999" LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999" fi -# check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS -if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then - echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in the RunPod cache." -fi - # check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error: if [[ "$LLAMA_SERVER_CMD_ARGS" == *"--port"* ]]; then echo "start.sh: Error: You must not define --port in LLAMA_SERVER_CMD_ARGS, as port 3098 is required." @@ -43,14 +54,14 @@ echo "start.sh: Stopping existing llama-server instances (if any)..." } # we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS; -# it contains a.e. "-hf modelname --ctx-size 4096 -ngl 99". +# it contains a.e. "-hf modelname --ctx-size 4096 -ngl 999". echo "start.sh: Running llama-server $LLAMA_SERVER_CMD_ARGS --port 3098" touch llama.server.log # We need to pass these arguments to llama-server verbatim. -LD_LIBRARY_PATH=/app /app/llama-server $LLAMA_SERVER_CMD_ARGS --port 3098 2>&1 | tee llama.server.log & +LD_LIBRARY_PATH=/app /app/llama-server $CACHED_LLAMA_ARGS $LLAMA_SERVER_CMD_ARGS --port 3098 2>&1 | tee llama.server.log & LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command