From 398b59c0a12b65a01aafc0dafca5ac6b86b91952 Mon Sep 17 00:00:00 2001 From: mags0ft Date: Thu, 25 Dec 2025 19:17:31 +0100 Subject: [PATCH] update test timeout and default LLAMA_SERVER_CMD_ARGS for improved performance --- .runpod/tests.json | 4 ++-- src/start.sh | 6 +++--- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/.runpod/tests.json b/.runpod/tests.json index c60272b..6e95318 100644 --- a/.runpod/tests.json +++ b/.runpod/tests.json @@ -5,7 +5,7 @@ "input": { "prompt": "Hi! Who are you?" }, - "timeout": 120000 + "timeout": 60000 } ], "config": { @@ -14,7 +14,7 @@ "env": [ { "key": "LLAMA_SERVER_CMD_ARGS", - "value": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096" + "value": "-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999" } ], "allowedCudaVersions": [ diff --git a/src/start.sh b/src/start.sh index db690ec..285ee0d 100644 --- a/src/start.sh +++ b/src/start.sh @@ -16,13 +16,13 @@ cleanup() { # check if $LLAMA_SERVER_CMD_ARGS is set if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then - echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096" - LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 99" + echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999" + LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999" fi # check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then - echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace." + echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in the RunPod cache." fi # check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error: