Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0a59d57025 | ||
|
|
398b59c0a1 | ||
|
|
889d3c5c96 | ||
|
|
41b3b3d02b | ||
|
|
a13d6c1b9f |
@@ -5,7 +5,7 @@
|
|||||||
"input": {
|
"input": {
|
||||||
"prompt": "Hi! Who are you?"
|
"prompt": "Hi! Who are you?"
|
||||||
},
|
},
|
||||||
"timeout": 120000
|
"timeout": 60000
|
||||||
}
|
}
|
||||||
],
|
],
|
||||||
"config": {
|
"config": {
|
||||||
@@ -14,7 +14,7 @@
|
|||||||
"env": [
|
"env": [
|
||||||
{
|
{
|
||||||
"key": "LLAMA_SERVER_CMD_ARGS",
|
"key": "LLAMA_SERVER_CMD_ARGS",
|
||||||
"value": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096"
|
"value": "-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
|
||||||
}
|
}
|
||||||
],
|
],
|
||||||
"allowedCudaVersions": [
|
"allowedCudaVersions": [
|
||||||
@@ -13,6 +13,10 @@ The following OpenAI API endpoints are supported:
|
|||||||
|
|
||||||
Streaming responses is also supported.
|
Streaming responses is also supported.
|
||||||
|
|
||||||
|
**Important!** This project is still relatively new. Please [open a new issue](https://github.com/Jacob-ML/inference-worker/issues/new) if you encounter any problems in order to get help.
|
||||||
|
|
||||||
|
**This is a fork of [SvenBrnn's `runpod-worker-ollama`](https://github.com/SvenBrnn/runpod-worker-ollama).**
|
||||||
|
|
||||||
## Setup
|
## Setup
|
||||||
|
|
||||||
For the setup to work best, it is recommended to use a network volume attached to all workers which stores the model GGUFs and then reference those files in the launch arguments.
|
For the setup to work best, it is recommended to use a network volume attached to all workers which stores the model GGUFs and then reference those files in the launch arguments.
|
||||||
|
|||||||
@@ -0,0 +1,59 @@
|
|||||||
|
"""
|
||||||
|
Finds the full LLM GGUF path from the Hugging Face cache.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import argparse
|
||||||
|
|
||||||
|
CACHE_DIR = "/runpod-volume/huggingface-cache/hub"
|
||||||
|
|
||||||
|
|
||||||
|
def find_model_path(model_name, gguf_in_repo="model.gguf"):
|
||||||
|
"""
|
||||||
|
Find the path to a cached model.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
model_name: The model name from Hugging Face
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The full path to the cached model, or None if not found
|
||||||
|
"""
|
||||||
|
|
||||||
|
cache_name = model_name.replace("/", "--")
|
||||||
|
snapshots_dir = os.path.join(
|
||||||
|
CACHE_DIR, f"models--{cache_name}", "snapshots"
|
||||||
|
)
|
||||||
|
|
||||||
|
if os.path.exists(snapshots_dir):
|
||||||
|
snapshots = os.listdir(snapshots_dir)
|
||||||
|
|
||||||
|
if snapshots:
|
||||||
|
return os.path.join(snapshots_dir, snapshots[0], gguf_in_repo)
|
||||||
|
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
"""
|
||||||
|
Main function to find and print the model path.
|
||||||
|
"""
|
||||||
|
|
||||||
|
parser = argparse.ArgumentParser(
|
||||||
|
description="Find the full GGUF path from the Hugging Face cache."
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"model", type=str, help="The model name from Hugging Face"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"path",
|
||||||
|
type=str,
|
||||||
|
help="The path to the GGUF file within the model repository",
|
||||||
|
)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
model_path = find_model_path(args.model, args.path)
|
||||||
|
print(model_path, end="")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
+3
-3
@@ -16,13 +16,13 @@ cleanup() {
|
|||||||
|
|
||||||
# check if $LLAMA_SERVER_CMD_ARGS is set
|
# check if $LLAMA_SERVER_CMD_ARGS is set
|
||||||
if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then
|
if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then
|
||||||
echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096"
|
echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
|
||||||
LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 99"
|
LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:IQ2_XXS --ctx-size 512 -ngl 999"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS
|
# check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS
|
||||||
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then
|
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then
|
||||||
echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace."
|
echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in the RunPod cache."
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error:
|
# check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error:
|
||||||
|
|||||||
Reference in New Issue
Block a user