6 Commits
5 changed files with 72 additions and 8 deletions
+2 -2
View File
@@ -5,7 +5,7 @@
"input": {
"prompt": "Hi! Who are you?"
},
"timeout": 120000
"timeout": 60000
}
],
"config": {
@@ -14,7 +14,7 @@
"env": [
{
"key": "LLAMA_SERVER_CMD_ARGS",
"value": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096"
"value": "-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
}
],
"allowedCudaVersions": [
+1 -1
View File
@@ -17,7 +17,7 @@
"name": "Command line arguments for llama-server",
"type": "string",
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.",
"default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096",
"default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 99",
"advanced": false
}
},
+4
View File
@@ -13,6 +13,10 @@ The following OpenAI API endpoints are supported:
Streaming responses is also supported.
**Important!** This project is still relatively new. Please [open a new issue](https://github.com/Jacob-ML/inference-worker/issues/new) if you encounter any problems in order to get help.
**This is a fork of [SvenBrnn's `runpod-worker-ollama`](https://github.com/SvenBrnn/runpod-worker-ollama).**
## Setup
For the setup to work best, it is recommended to use a network volume attached to all workers which stores the model GGUFs and then reference those files in the launch arguments.
+59
View File
@@ -0,0 +1,59 @@
"""
Finds the full LLM GGUF path from the Hugging Face cache.
"""
import os
import argparse
CACHE_DIR = "/runpod-volume/huggingface-cache/hub"
def find_model_path(model_name, gguf_in_repo="model.gguf"):
"""
Find the path to a cached model.
Args:
model_name: The model name from Hugging Face
Returns:
The full path to the cached model, or None if not found
"""
cache_name = model_name.replace("/", "--")
snapshots_dir = os.path.join(
CACHE_DIR, f"models--{cache_name}", "snapshots"
)
if os.path.exists(snapshots_dir):
snapshots = os.listdir(snapshots_dir)
if snapshots:
return os.path.join(snapshots_dir, snapshots[0], gguf_in_repo)
return None
def main():
"""
Main function to find and print the model path.
"""
parser = argparse.ArgumentParser(
description="Find the full GGUF path from the Hugging Face cache."
)
parser.add_argument(
"model", type=str, help="The model name from Hugging Face"
)
parser.add_argument(
"path",
type=str,
help="The path to the GGUF file within the model repository",
)
args = parser.parse_args()
model_path = find_model_path(args.model, args.path)
print(model_path, end="")
if __name__ == "__main__":
main()
+6 -5
View File
@@ -16,13 +16,13 @@ cleanup() {
# check if $LLAMA_SERVER_CMD_ARGS is set
if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then
echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096"
LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096"
echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
fi
# check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then
echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace."
echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in the RunPod cache."
fi
# check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error:
@@ -43,7 +43,7 @@ echo "start.sh: Stopping existing llama-server instances (if any)..."
}
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
# it contains a.e. "-hf modelname --ctx-size 4096".
# it contains a.e. "-hf modelname --ctx-size 4096 -ngl 99".
echo "start.sh: Running llama-server $LLAMA_SERVER_CMD_ARGS --port 3098"
@@ -68,7 +68,8 @@ echo "start.sh: Waiting for llama-server to start..."
# wait for the server to start
while ! check_server_is_running; do
sleep 5
# we don't want to lose too much time, so we check very frequently
sleep 0.5
done
echo "start.sh: llama-server is up and running, delegating to the handler script."