Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f890c50c8a | ||
|
|
dd2d52b5e3 | ||
|
|
1196621d77 | ||
|
|
3bcbee2d93 | ||
|
|
b24c32c024 | ||
|
|
c7b115bec0 | ||
|
|
8a2982faa4 | ||
|
|
d4ac091735 |
+1
-1
@@ -13,7 +13,7 @@ However, this will cause every worker to download the model from the Hugging Fac
|
|||||||
A naive way to cache the model would be to store it on a network volume in RunPod and reference the model files this way:
|
A naive way to cache the model would be to store it on a network volume in RunPod and reference the model files this way:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
-hf /runpod-volume/model.gguf --ctx-size 4096 # etc...
|
-m /runpod-volume/model.gguf --ctx-size 4096 # etc...
|
||||||
```
|
```
|
||||||
|
|
||||||
Unfortunately, network volume performance is often not sufficient for loading large models, leading to long load times. RunPod introduced a [caching mechanism](https://docs.runpod.io/serverless/endpoints/model-caching) to solve this problem.
|
Unfortunately, network volume performance is often not sufficient for loading large models, leading to long load times. RunPod introduced a [caching mechanism](https://docs.runpod.io/serverless/endpoints/model-caching) to solve this problem.
|
||||||
|
|||||||
+8
-1
@@ -3,6 +3,7 @@ Finds the full LLM GGUF path from the Hugging Face cache.
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
import os
|
import os
|
||||||
|
import sys
|
||||||
import argparse
|
import argparse
|
||||||
|
|
||||||
CACHE_DIR = "/runpod-volume/huggingface-cache/hub"
|
CACHE_DIR = "/runpod-volume/huggingface-cache/hub"
|
||||||
@@ -19,7 +20,7 @@ def find_model_path(model_name, gguf_in_repo="model.gguf"):
|
|||||||
The full path to the cached model, or None if not found
|
The full path to the cached model, or None if not found
|
||||||
"""
|
"""
|
||||||
|
|
||||||
cache_name = model_name.replace("/", "--")
|
cache_name = model_name.replace("/", "--").lower()
|
||||||
snapshots_dir = os.path.join(
|
snapshots_dir = os.path.join(
|
||||||
CACHE_DIR, f"models--{cache_name}", "snapshots"
|
CACHE_DIR, f"models--{cache_name}", "snapshots"
|
||||||
)
|
)
|
||||||
@@ -52,6 +53,12 @@ def main():
|
|||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
model_path = find_model_path(args.model, args.path)
|
model_path = find_model_path(args.model, args.path)
|
||||||
|
if model_path is None:
|
||||||
|
print(
|
||||||
|
f"Error: Cached model not found. Model='{args.model}', GGUF='{args.path}', Cache dir='{CACHE_DIR}'",
|
||||||
|
file=sys.stderr,
|
||||||
|
)
|
||||||
|
sys.exit(1)
|
||||||
print(model_path, end="")
|
print(model_path, end="")
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+17
-2
@@ -17,7 +17,13 @@ cleanup() {
|
|||||||
CACHED_LLAMA_ARGS=""
|
CACHED_LLAMA_ARGS=""
|
||||||
|
|
||||||
find_cached_path() {
|
find_cached_path() {
|
||||||
CACHED_LLAMA_ARGS="-m $(python ./find_cached.py $LLAMA_CACHED_MODEL $LLAMA_CACHED_GGUF_PATH)"
|
local model_path
|
||||||
|
model_path=$(python ./find_cached.py "$LLAMA_CACHED_MODEL" "$LLAMA_CACHED_GGUF_PATH")
|
||||||
|
if [ $? -ne 0 ] || [ -z "$model_path" ]; then
|
||||||
|
echo "start.sh: Error: Could not resolve cached model path. Check that LLAMA_CACHED_MODEL and LLAMA_CACHED_GGUF_PATH are correct and the model is fully cached on the network volume."
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
CACHED_LLAMA_ARGS="-m $model_path"
|
||||||
}
|
}
|
||||||
|
|
||||||
# check if $LLAMA_CACHED_MODEL is set and not empty
|
# check if $LLAMA_CACHED_MODEL is set and not empty
|
||||||
@@ -56,7 +62,7 @@ echo "start.sh: Stopping existing llama-server instances (if any)..."
|
|||||||
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
|
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
|
||||||
# it contains a.e. "-hf modelname --ctx-size 4096 -ngl 999".
|
# it contains a.e. "-hf modelname --ctx-size 4096 -ngl 999".
|
||||||
|
|
||||||
echo "start.sh: Running llama-server $LLAMA_SERVER_CMD_ARGS --port 3098"
|
echo "start.sh: Running /app/llama-server $CACHED_LLAMA_ARGS $LLAMA_SERVER_CMD_ARGS --port 3098"
|
||||||
|
|
||||||
touch llama.server.log
|
touch llama.server.log
|
||||||
|
|
||||||
@@ -65,6 +71,8 @@ LD_LIBRARY_PATH=/app /app/llama-server $CACHED_LLAMA_ARGS $LLAMA_SERVER_CMD_ARGS
|
|||||||
|
|
||||||
LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command
|
LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command
|
||||||
|
|
||||||
|
tries_so_far=0
|
||||||
|
|
||||||
check_server_is_running() {
|
check_server_is_running() {
|
||||||
echo "start.sh: Checking if llama-server is done initializing..."
|
echo "start.sh: Checking if llama-server is done initializing..."
|
||||||
|
|
||||||
@@ -74,6 +82,13 @@ check_server_is_running() {
|
|||||||
return 1 # failure
|
return 1 # failure
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
tries_so_far=$((tries_so_far + 1))
|
||||||
|
|
||||||
|
if [ $tries_so_far -ge 120 ]; then
|
||||||
|
echo "start.sh: Error: llama-server did not start within 60 seconds."
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
# check if the process is still running
|
# check if the process is still running
|
||||||
if ! kill -0 $LLAMA_SERVER_PID 2>/dev/null; then
|
if ! kill -0 $LLAMA_SERVER_PID 2>/dev/null; then
|
||||||
echo "start.sh: Error: llama-server process has exited unexpectedly."
|
echo "start.sh: Error: llama-server process has exited unexpectedly."
|
||||||
|
|||||||
Reference in New Issue
Block a user