5 Commits
Author SHA1 Message Date
mags0ft 0a59d57025 rename tests.json temporarily to skip apparently buggy RunPod CI
The RunPod support told me to do this until the issues are resolved.
2025-12-26 11:37:23 +01:00
mags0ft 398b59c0a1 update test timeout and default LLAMA_SERVER_CMD_ARGS for improved performance 2025-12-25 19:17:31 +01:00
mags0ft 889d3c5c96 move find_cached.py to src 2025-12-18 10:57:10 +01:00
mags0ft 41b3b3d02b add helper script to use cached models 2025-12-18 10:50:33 +01:00
mags0ftandGitHub a13d6c1b9f add issue note and fork explanation to README 2025-11-24 22:24:46 +01:00
4 changed files with 68 additions and 5 deletions
+2 -2
View File
@@ -5,7 +5,7 @@
"input": { "input": {
"prompt": "Hi! Who are you?" "prompt": "Hi! Who are you?"
}, },
"timeout": 120000 "timeout": 60000
} }
], ],
"config": { "config": {
@@ -14,7 +14,7 @@
"env": [ "env": [
{ {
"key": "LLAMA_SERVER_CMD_ARGS", "key": "LLAMA_SERVER_CMD_ARGS",
"value": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096" "value": "-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
} }
], ],
"allowedCudaVersions": [ "allowedCudaVersions": [
+4
View File
@@ -13,6 +13,10 @@ The following OpenAI API endpoints are supported:
Streaming responses is also supported. Streaming responses is also supported.
**Important!** This project is still relatively new. Please [open a new issue](https://github.com/Jacob-ML/inference-worker/issues/new) if you encounter any problems in order to get help.
**This is a fork of [SvenBrnn's `runpod-worker-ollama`](https://github.com/SvenBrnn/runpod-worker-ollama).**
## Setup ## Setup
For the setup to work best, it is recommended to use a network volume attached to all workers which stores the model GGUFs and then reference those files in the launch arguments. For the setup to work best, it is recommended to use a network volume attached to all workers which stores the model GGUFs and then reference those files in the launch arguments.
+59
View File
@@ -0,0 +1,59 @@
"""
Finds the full LLM GGUF path from the Hugging Face cache.
"""
import os
import argparse
CACHE_DIR = "/runpod-volume/huggingface-cache/hub"
def find_model_path(model_name, gguf_in_repo="model.gguf"):
"""
Find the path to a cached model.
Args:
model_name: The model name from Hugging Face
Returns:
The full path to the cached model, or None if not found
"""
cache_name = model_name.replace("/", "--")
snapshots_dir = os.path.join(
CACHE_DIR, f"models--{cache_name}", "snapshots"
)
if os.path.exists(snapshots_dir):
snapshots = os.listdir(snapshots_dir)
if snapshots:
return os.path.join(snapshots_dir, snapshots[0], gguf_in_repo)
return None
def main():
"""
Main function to find and print the model path.
"""
parser = argparse.ArgumentParser(
description="Find the full GGUF path from the Hugging Face cache."
)
parser.add_argument(
"model", type=str, help="The model name from Hugging Face"
)
parser.add_argument(
"path",
type=str,
help="The path to the GGUF file within the model repository",
)
args = parser.parse_args()
model_path = find_model_path(args.model, args.path)
print(model_path, end="")
if __name__ == "__main__":
main()
+3 -3
View File
@@ -16,13 +16,13 @@ cleanup() {
# check if $LLAMA_SERVER_CMD_ARGS is set # check if $LLAMA_SERVER_CMD_ARGS is set
if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then
echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096" echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 99" LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
fi fi
# check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS # check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then
echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace." echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in the RunPod cache."
fi fi
# check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error: # check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error: