4 Commits
3 changed files with 11 additions and 2 deletions
+1 -1
View File
@@ -13,7 +13,7 @@ However, this will cause every worker to download the model from the Hugging Fac
A naive way to cache the model would be to store it on a network volume in RunPod and reference the model files this way: A naive way to cache the model would be to store it on a network volume in RunPod and reference the model files this way:
```bash ```bash
-hf /runpod-volume/model.gguf --ctx-size 4096 # etc... -m /runpod-volume/model.gguf --ctx-size 4096 # etc...
``` ```
Unfortunately, network volume performance is often not sufficient for loading large models, leading to long load times. RunPod introduced a [caching mechanism](https://docs.runpod.io/serverless/endpoints/model-caching) to solve this problem. Unfortunately, network volume performance is often not sufficient for loading large models, leading to long load times. RunPod introduced a [caching mechanism](https://docs.runpod.io/serverless/endpoints/model-caching) to solve this problem.
+10 -1
View File
@@ -56,7 +56,7 @@ echo "start.sh: Stopping existing llama-server instances (if any)..."
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS; # we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
# it contains a.e. "-hf modelname --ctx-size 4096 -ngl 999". # it contains a.e. "-hf modelname --ctx-size 4096 -ngl 999".
echo "start.sh: Running llama-server $LLAMA_SERVER_CMD_ARGS --port 3098" echo "start.sh: Running /app/llama-server $CACHED_LLAMA_ARGS $LLAMA_SERVER_CMD_ARGS --port 3098"
touch llama.server.log touch llama.server.log
@@ -65,6 +65,8 @@ LD_LIBRARY_PATH=/app /app/llama-server $CACHED_LLAMA_ARGS $LLAMA_SERVER_CMD_ARGS
LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command
tries_so_far=0
check_server_is_running() { check_server_is_running() {
echo "start.sh: Checking if llama-server is done initializing..." echo "start.sh: Checking if llama-server is done initializing..."
@@ -74,6 +76,13 @@ check_server_is_running() {
return 1 # failure return 1 # failure
fi fi
tries_so_far=$((tries_so_far + 1))
if [ $tries_so_far -ge 120 ]; then
echo "start.sh: Error: llama-server did not start within 60 seconds."
exit 1
fi
# check if the process is still running # check if the process is still running
if ! kill -0 $LLAMA_SERVER_PID 2>/dev/null; then if ! kill -0 $LLAMA_SERVER_PID 2>/dev/null; then
echo "start.sh: Error: llama-server process has exited unexpectedly." echo "start.sh: Error: llama-server process has exited unexpectedly."