Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
029f37afab | ||
|
|
85c10953d6 | ||
|
|
ea82e2a901 | ||
|
|
f890c50c8a | ||
|
|
dd2d52b5e3 | ||
|
|
1196621d77 | ||
|
|
3bcbee2d93 | ||
|
|
b24c32c024 | ||
|
|
c7b115bec0 |
+1
-1
@@ -1,5 +1,5 @@
|
|||||||
{
|
{
|
||||||
"title": "llama.cpp inference",
|
"title": "llama.cpp",
|
||||||
"description": "Run llama.cpp inference using serverless RunPod workers!",
|
"description": "Run llama.cpp inference using serverless RunPod workers!",
|
||||||
"type": "serverless",
|
"type": "serverless",
|
||||||
"category": "language",
|
"category": "language",
|
||||||
|
|||||||
+1
-1
@@ -13,7 +13,7 @@ However, this will cause every worker to download the model from the Hugging Fac
|
|||||||
A naive way to cache the model would be to store it on a network volume in RunPod and reference the model files this way:
|
A naive way to cache the model would be to store it on a network volume in RunPod and reference the model files this way:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
-hf /runpod-volume/model.gguf --ctx-size 4096 # etc...
|
-m /runpod-volume/model.gguf --ctx-size 4096 # etc...
|
||||||
```
|
```
|
||||||
|
|
||||||
Unfortunately, network volume performance is often not sufficient for loading large models, leading to long load times. RunPod introduced a [caching mechanism](https://docs.runpod.io/serverless/endpoints/model-caching) to solve this problem.
|
Unfortunately, network volume performance is often not sufficient for loading large models, leading to long load times. RunPod introduced a [caching mechanism](https://docs.runpod.io/serverless/endpoints/model-caching) to solve this problem.
|
||||||
|
|||||||
+1
-1
@@ -28,7 +28,7 @@ from utils import JobInput
|
|||||||
|
|
||||||
client = OpenAI(
|
client = OpenAI(
|
||||||
base_url="http://localhost:3098/v1/",
|
base_url="http://localhost:3098/v1/",
|
||||||
api_key="",
|
api_key="none",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+28
-33
@@ -1,69 +1,64 @@
|
|||||||
"""
|
"""
|
||||||
Finds the full GGUF path from the Hugging Face cache using huggingface_hub.
|
Finds the full LLM GGUF path from the Hugging Face cache.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import argparse
|
|
||||||
import os
|
import os
|
||||||
from huggingface_hub import snapshot_download, scan_cache_dir
|
import sys
|
||||||
|
import argparse
|
||||||
|
|
||||||
|
CACHE_DIR = "/runpod-volume/huggingface-cache/hub"
|
||||||
|
|
||||||
|
|
||||||
def find_model_path(model_name: str, gguf_in_repo: str) -> str | None:
|
def find_model_path(model_name, gguf_in_repo="model.gguf"):
|
||||||
"""
|
"""
|
||||||
Resolve the GGUF file path from the Hugging Face cache.
|
Find the path to a cached model.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
model_name: Hugging Face repo id (e.g. TheBloke/Mistral-7B-GGUF)
|
model_name: The model name from Hugging Face
|
||||||
gguf_in_repo: Relative path to the GGUF file inside the repo
|
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Full filesystem path to the GGUF file, or None if not found
|
The full path to the cached model, or None if not found
|
||||||
"""
|
"""
|
||||||
|
|
||||||
# First try the official way: resolve snapshot path from cache
|
cache_name = model_name.replace("/", "--").lower()
|
||||||
try:
|
snapshots_dir = os.path.join(
|
||||||
snapshot_path = snapshot_download(
|
CACHE_DIR, f"models--{cache_name}", "snapshots"
|
||||||
repo_id=model_name,
|
|
||||||
local_files_only=True
|
|
||||||
)
|
)
|
||||||
candidate = os.path.join(snapshot_path, gguf_in_repo)
|
|
||||||
if os.path.isfile(candidate):
|
|
||||||
return candidate
|
|
||||||
except Exception:
|
|
||||||
pass
|
|
||||||
|
|
||||||
# Fallback: scan cache metadata explicitly (no downloads)
|
if os.path.exists(snapshots_dir):
|
||||||
cache = scan_cache_dir()
|
snapshots = os.listdir(snapshots_dir)
|
||||||
for repo in cache.repos:
|
|
||||||
if repo.repo_id == model_name:
|
if snapshots:
|
||||||
for revision in repo.revisions:
|
return os.path.join(snapshots_dir, snapshots[0], gguf_in_repo)
|
||||||
candidate = os.path.join(revision.snapshot_path, gguf_in_repo)
|
|
||||||
if os.path.isfile(candidate):
|
|
||||||
return candidate
|
|
||||||
|
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
|
"""
|
||||||
|
Main function to find and print the model path.
|
||||||
|
"""
|
||||||
|
|
||||||
parser = argparse.ArgumentParser(
|
parser = argparse.ArgumentParser(
|
||||||
description="Find the full GGUF path from the Hugging Face cache."
|
description="Find the full GGUF path from the Hugging Face cache."
|
||||||
)
|
)
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"model",
|
"model", type=str, help="The model name from Hugging Face"
|
||||||
type=str,
|
|
||||||
help="Hugging Face model repo id (e.g. TheBloke/Mistral-7B-GGUF)",
|
|
||||||
)
|
)
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"path",
|
"path",
|
||||||
type=str,
|
type=str,
|
||||||
help="Relative path to the GGUF file inside the repo",
|
help="The path to the GGUF file within the model repository",
|
||||||
)
|
)
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
model_path = find_model_path(args.model, args.path)
|
model_path = find_model_path(args.model, args.path)
|
||||||
|
|
||||||
if model_path is None:
|
if model_path is None:
|
||||||
raise SystemExit("GGUF file not found in Hugging Face cache")
|
print(
|
||||||
|
f"Error: Cached model not found. Model='{args.model}', GGUF='{args.path}', Cache dir='{CACHE_DIR}'",
|
||||||
|
file=sys.stderr,
|
||||||
|
)
|
||||||
|
sys.exit(1)
|
||||||
print(model_path, end="")
|
print(model_path, end="")
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +1,3 @@
|
|||||||
runpod
|
runpod==1.9.1
|
||||||
python-dotenv
|
python-dotenv==1.2.2
|
||||||
openai
|
openai==2.41.1
|
||||||
huggingface_hub
|
|
||||||
hf-transfer
|
|
||||||
|
|||||||
+7
-1
@@ -17,7 +17,13 @@ cleanup() {
|
|||||||
CACHED_LLAMA_ARGS=""
|
CACHED_LLAMA_ARGS=""
|
||||||
|
|
||||||
find_cached_path() {
|
find_cached_path() {
|
||||||
CACHED_LLAMA_ARGS="-m $(python ./find_cached.py $LLAMA_CACHED_MODEL $LLAMA_CACHED_GGUF_PATH)"
|
local model_path
|
||||||
|
model_path=$(python ./find_cached.py "$LLAMA_CACHED_MODEL" "$LLAMA_CACHED_GGUF_PATH")
|
||||||
|
if [ $? -ne 0 ] || [ -z "$model_path" ]; then
|
||||||
|
echo "start.sh: Error: Could not resolve cached model path. Check that LLAMA_CACHED_MODEL and LLAMA_CACHED_GGUF_PATH are correct and the model is fully cached on the network volume."
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
CACHED_LLAMA_ARGS="-m $model_path"
|
||||||
}
|
}
|
||||||
|
|
||||||
# check if $LLAMA_CACHED_MODEL is set and not empty
|
# check if $LLAMA_CACHED_MODEL is set and not empty
|
||||||
|
|||||||
Reference in New Issue
Block a user