update default args to include -ngl 99, improve startup sleep duration

This commit is contained in:
mags0ft
2025-11-19 19:16:18 +01:00
parent 381ed9ffff
commit 403d318ffc
2 changed files with 5 additions and 4 deletions
+1 -1
View File
@@ -17,7 +17,7 @@
"name": "Command line arguments for llama-server", "name": "Command line arguments for llama-server",
"type": "string", "type": "string",
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.", "description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.",
"default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096", "default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 99",
"advanced": false "advanced": false
} }
}, },
+4 -3
View File
@@ -17,7 +17,7 @@ cleanup() {
# check if $LLAMA_SERVER_CMD_ARGS is set # check if $LLAMA_SERVER_CMD_ARGS is set
if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then
echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096" echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096"
LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096" LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 99"
fi fi
# check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS # check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS
@@ -43,7 +43,7 @@ echo "start.sh: Stopping existing llama-server instances (if any)..."
} }
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS; # we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
# it contains a.e. "-hf modelname --ctx-size 4096". # it contains a.e. "-hf modelname --ctx-size 4096 -ngl 99".
echo "start.sh: Running llama-server $LLAMA_SERVER_CMD_ARGS --port 3098" echo "start.sh: Running llama-server $LLAMA_SERVER_CMD_ARGS --port 3098"
@@ -68,7 +68,8 @@ echo "start.sh: Waiting for llama-server to start..."
# wait for the server to start # wait for the server to start
while ! check_server_is_running; do while ! check_server_is_running; do
sleep 5 # we don't want to lose too much time, so we check very frequently
sleep 0.5
done done
echo "start.sh: llama-server is up and running, delegating to the handler script." echo "start.sh: llama-server is up and running, delegating to the handler script."