11 Commits
Author SHA1 Message Date
MagnusandGitHub 029f37afab pin dependencies 2026-06-14 18:03:54 +02:00
MagnusandGitHub 85c10953d6 add pseudo API key so the client doesn't raise an error
this seems to have been introduced in a recent OpenAI SDK update
2026-06-14 18:00:42 +02:00
MagnusandGitHub ea82e2a901 remove "inference" from title to make it more consistent with the other runpod repos on the hub 2026-06-03 18:22:06 +02:00
MagnusandGitHub f890c50c8a Merge pull request #3 from cardinalfan1/claude/fix-runpod-startup-xxJqT
Fix -m None passed to llama-server when cached model not found
2026-06-03 18:12:26 +02:00
Claude dd2d52b5e3 Fix -m None passed to llama-server when cached model not found
When LLAMA_CACHED_MODEL is set but the model isn't present in the cache,
find_cached.py was printing Python's None as the string "None", causing
start.sh to pass "-m None" to llama-server.

- find_cached.py: print an error to stderr and exit 1 when the model path
  cannot be resolved, instead of printing "None"
- start.sh: capture find_cached.py output into a local variable, check
  the exit code and guard against empty output before constructing
  CACHED_LLAMA_ARGS; also quote the env-var expansions to handle spaces

https://claude.ai/code/session_011ny5CFYnrzPbRneSzio5CR
2026-04-15 19:34:58 +00:00
mags0ftandGitHub 1196621d77 Merge pull request #1 from aparmar2000/patch-1
automatically convert path to lowercase
2026-03-10 19:11:04 +01:00
Ashan ParmarandGitHub 3bcbee2d93 Update find_cached.py
The cache name is lowercase
2026-02-20 18:04:59 -07:00
mags0ft b24c32c024 re-activate automatic testing 2026-02-12 08:45:54 +01:00
mags0ftandGitHub c7b115bec0 fix typo in cached models docs 2026-02-11 02:14:18 +01:00
mags0ft 8a2982faa4 adjust incorrect llama-server start command log 2025-12-27 23:51:00 +01:00
mags0ft d4ac091735 add timeout for llama-server startup to prevent indefinite waiting 2025-12-27 18:00:58 +01:00
7 changed files with 32 additions and 10 deletions
+2 -2
View File
@@ -1,5 +1,5 @@
{ {
"title": "llama.cpp inference", "title": "llama.cpp",
"description": "Run llama.cpp inference using serverless RunPod workers!", "description": "Run llama.cpp inference using serverless RunPod workers!",
"type": "serverless", "type": "serverless",
"category": "language", "category": "language",
@@ -53,4 +53,4 @@
} }
] ]
} }
} }
+1 -1
View File
@@ -13,7 +13,7 @@ However, this will cause every worker to download the model from the Hugging Fac
A naive way to cache the model would be to store it on a network volume in RunPod and reference the model files this way: A naive way to cache the model would be to store it on a network volume in RunPod and reference the model files this way:
```bash ```bash
-hf /runpod-volume/model.gguf --ctx-size 4096 # etc... -m /runpod-volume/model.gguf --ctx-size 4096 # etc...
``` ```
Unfortunately, network volume performance is often not sufficient for loading large models, leading to long load times. RunPod introduced a [caching mechanism](https://docs.runpod.io/serverless/endpoints/model-caching) to solve this problem. Unfortunately, network volume performance is often not sufficient for loading large models, leading to long load times. RunPod introduced a [caching mechanism](https://docs.runpod.io/serverless/endpoints/model-caching) to solve this problem.
+1 -1
View File
@@ -28,7 +28,7 @@ from utils import JobInput
client = OpenAI( client = OpenAI(
base_url="http://localhost:3098/v1/", base_url="http://localhost:3098/v1/",
api_key="", api_key="none",
) )
+8 -1
View File
@@ -3,6 +3,7 @@ Finds the full LLM GGUF path from the Hugging Face cache.
""" """
import os import os
import sys
import argparse import argparse
CACHE_DIR = "/runpod-volume/huggingface-cache/hub" CACHE_DIR = "/runpod-volume/huggingface-cache/hub"
@@ -19,7 +20,7 @@ def find_model_path(model_name, gguf_in_repo="model.gguf"):
The full path to the cached model, or None if not found The full path to the cached model, or None if not found
""" """
cache_name = model_name.replace("/", "--") cache_name = model_name.replace("/", "--").lower()
snapshots_dir = os.path.join( snapshots_dir = os.path.join(
CACHE_DIR, f"models--{cache_name}", "snapshots" CACHE_DIR, f"models--{cache_name}", "snapshots"
) )
@@ -52,6 +53,12 @@ def main():
args = parser.parse_args() args = parser.parse_args()
model_path = find_model_path(args.model, args.path) model_path = find_model_path(args.model, args.path)
if model_path is None:
print(
f"Error: Cached model not found. Model='{args.model}', GGUF='{args.path}', Cache dir='{CACHE_DIR}'",
file=sys.stderr,
)
sys.exit(1)
print(model_path, end="") print(model_path, end="")
+3 -3
View File
@@ -1,3 +1,3 @@
runpod runpod==1.9.1
python-dotenv python-dotenv==1.2.2
openai openai==2.41.1
+17 -2
View File
@@ -17,7 +17,13 @@ cleanup() {
CACHED_LLAMA_ARGS="" CACHED_LLAMA_ARGS=""
find_cached_path() { find_cached_path() {
CACHED_LLAMA_ARGS="-m $(python ./find_cached.py $LLAMA_CACHED_MODEL $LLAMA_CACHED_GGUF_PATH)" local model_path
model_path=$(python ./find_cached.py "$LLAMA_CACHED_MODEL" "$LLAMA_CACHED_GGUF_PATH")
if [ $? -ne 0 ] || [ -z "$model_path" ]; then
echo "start.sh: Error: Could not resolve cached model path. Check that LLAMA_CACHED_MODEL and LLAMA_CACHED_GGUF_PATH are correct and the model is fully cached on the network volume."
exit 1
fi
CACHED_LLAMA_ARGS="-m $model_path"
} }
# check if $LLAMA_CACHED_MODEL is set and not empty # check if $LLAMA_CACHED_MODEL is set and not empty
@@ -56,7 +62,7 @@ echo "start.sh: Stopping existing llama-server instances (if any)..."
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS; # we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
# it contains a.e. "-hf modelname --ctx-size 4096 -ngl 999". # it contains a.e. "-hf modelname --ctx-size 4096 -ngl 999".
echo "start.sh: Running llama-server $LLAMA_SERVER_CMD_ARGS --port 3098" echo "start.sh: Running /app/llama-server $CACHED_LLAMA_ARGS $LLAMA_SERVER_CMD_ARGS --port 3098"
touch llama.server.log touch llama.server.log
@@ -65,6 +71,8 @@ LD_LIBRARY_PATH=/app /app/llama-server $CACHED_LLAMA_ARGS $LLAMA_SERVER_CMD_ARGS
LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command
tries_so_far=0
check_server_is_running() { check_server_is_running() {
echo "start.sh: Checking if llama-server is done initializing..." echo "start.sh: Checking if llama-server is done initializing..."
@@ -74,6 +82,13 @@ check_server_is_running() {
return 1 # failure return 1 # failure
fi fi
tries_so_far=$((tries_so_far + 1))
if [ $tries_so_far -ge 120 ]; then
echo "start.sh: Error: llama-server did not start within 60 seconds."
exit 1
fi
# check if the process is still running # check if the process is still running
if ! kill -0 $LLAMA_SERVER_PID 2>/dev/null; then if ! kill -0 $LLAMA_SERVER_PID 2>/dev/null; then
echo "start.sh: Error: llama-server process has exited unexpectedly." echo "start.sh: Error: llama-server process has exited unexpectedly."