8 Commits
6 changed files with 77 additions and 8 deletions
+2 -2
View File
@@ -5,7 +5,7 @@
"input": { "input": {
"prompt": "Hi! Who are you?" "prompt": "Hi! Who are you?"
}, },
"timeout": 120000 "timeout": 60000
} }
], ],
"config": { "config": {
@@ -14,7 +14,7 @@
"env": [ "env": [
{ {
"key": "LLAMA_SERVER_CMD_ARGS", "key": "LLAMA_SERVER_CMD_ARGS",
"value": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096" "value": "-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
} }
], ],
"allowedCudaVersions": [ "allowedCudaVersions": [
+2 -2
View File
@@ -14,10 +14,10 @@
{ {
"key": "LLAMA_SERVER_CMD_ARGS", "key": "LLAMA_SERVER_CMD_ARGS",
"input": { "input": {
"name": "Model Name", "name": "Command line arguments for llama-server",
"type": "string", "type": "string",
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.", "description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.",
"default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096", "default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 99",
"advanced": false "advanced": false
} }
}, },
+4
View File
@@ -13,6 +13,10 @@ The following OpenAI API endpoints are supported:
Streaming responses is also supported. Streaming responses is also supported.
**Important!** This project is still relatively new. Please [open a new issue](https://github.com/Jacob-ML/inference-worker/issues/new) if you encounter any problems in order to get help.
**This is a fork of [SvenBrnn's `runpod-worker-ollama`](https://github.com/SvenBrnn/runpod-worker-ollama).**
## Setup ## Setup
For the setup to work best, it is recommended to use a network volume attached to all workers which stores the model GGUFs and then reference those files in the launch arguments. For the setup to work best, it is recommended to use a network volume attached to all workers which stores the model GGUFs and then reference those files in the launch arguments.
-1
View File
@@ -21,7 +21,6 @@ Typical usage:
""" """
import json import json
import os
from dotenv import load_dotenv from dotenv import load_dotenv
from openai import OpenAI from openai import OpenAI
+59
View File
@@ -0,0 +1,59 @@
"""
Finds the full LLM GGUF path from the Hugging Face cache.
"""
import os
import argparse
CACHE_DIR = "/runpod-volume/huggingface-cache/hub"
def find_model_path(model_name, gguf_in_repo="model.gguf"):
"""
Find the path to a cached model.
Args:
model_name: The model name from Hugging Face
Returns:
The full path to the cached model, or None if not found
"""
cache_name = model_name.replace("/", "--")
snapshots_dir = os.path.join(
CACHE_DIR, f"models--{cache_name}", "snapshots"
)
if os.path.exists(snapshots_dir):
snapshots = os.listdir(snapshots_dir)
if snapshots:
return os.path.join(snapshots_dir, snapshots[0], gguf_in_repo)
return None
def main():
"""
Main function to find and print the model path.
"""
parser = argparse.ArgumentParser(
description="Find the full GGUF path from the Hugging Face cache."
)
parser.add_argument(
"model", type=str, help="The model name from Hugging Face"
)
parser.add_argument(
"path",
type=str,
help="The path to the GGUF file within the model repository",
)
args = parser.parse_args()
model_path = find_model_path(args.model, args.path)
print(model_path, end="")
if __name__ == "__main__":
main()
+10 -3
View File
@@ -14,9 +14,15 @@ cleanup() {
exit 0 exit 0
} }
# check if $LLAMA_SERVER_CMD_ARGS is set
if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then
echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
fi
# check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS # check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then
echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace." echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in the RunPod cache."
fi fi
# check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error: # check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error:
@@ -37,7 +43,7 @@ echo "start.sh: Stopping existing llama-server instances (if any)..."
} }
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS; # we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
# it contains a.e. "-hf modelname --ctx-size 4096". # it contains a.e. "-hf modelname --ctx-size 4096 -ngl 99".
echo "start.sh: Running llama-server $LLAMA_SERVER_CMD_ARGS --port 3098" echo "start.sh: Running llama-server $LLAMA_SERVER_CMD_ARGS --port 3098"
@@ -62,7 +68,8 @@ echo "start.sh: Waiting for llama-server to start..."
# wait for the server to start # wait for the server to start
while ! check_server_is_running; do while ! check_server_is_running; do
sleep 5 # we don't want to lose too much time, so we check very frequently
sleep 0.5
done done
echo "start.sh: llama-server is up and running, delegating to the handler script." echo "start.sh: llama-server is up and running, delegating to the handler script."