2 Commits
5 changed files with 38 additions and 37 deletions
+1 -1
View File
@@ -13,7 +13,7 @@ However, this will cause every worker to download the model from the Hugging Fac
A naive way to cache the model would be to store it on a network volume in RunPod and reference the model files this way: A naive way to cache the model would be to store it on a network volume in RunPod and reference the model files this way:
```bash ```bash
-m /runpod-volume/model.gguf --ctx-size 4096 # etc... -hf /runpod-volume/model.gguf --ctx-size 4096 # etc...
``` ```
Unfortunately, network volume performance is often not sufficient for loading large models, leading to long load times. RunPod introduced a [caching mechanism](https://docs.runpod.io/serverless/endpoints/model-caching) to solve this problem. Unfortunately, network volume performance is often not sufficient for loading large models, leading to long load times. RunPod introduced a [caching mechanism](https://docs.runpod.io/serverless/endpoints/model-caching) to solve this problem.
+33 -28
View File
@@ -1,64 +1,69 @@
""" """
Finds the full LLM GGUF path from the Hugging Face cache. Finds the full GGUF path from the Hugging Face cache using huggingface_hub.
""" """
import os
import sys
import argparse import argparse
import os
CACHE_DIR = "/runpod-volume/huggingface-cache/hub" from huggingface_hub import snapshot_download, scan_cache_dir
def find_model_path(model_name, gguf_in_repo="model.gguf"): def find_model_path(model_name: str, gguf_in_repo: str) -> str | None:
""" """
Find the path to a cached model. Resolve the GGUF file path from the Hugging Face cache.
Args: Args:
model_name: The model name from Hugging Face model_name: Hugging Face repo id (e.g. TheBloke/Mistral-7B-GGUF)
gguf_in_repo: Relative path to the GGUF file inside the repo
Returns: Returns:
The full path to the cached model, or None if not found Full filesystem path to the GGUF file, or None if not found
""" """
cache_name = model_name.replace("/", "--").lower() # First try the official way: resolve snapshot path from cache
snapshots_dir = os.path.join( try:
CACHE_DIR, f"models--{cache_name}", "snapshots" snapshot_path = snapshot_download(
repo_id=model_name,
local_files_only=True
) )
candidate = os.path.join(snapshot_path, gguf_in_repo)
if os.path.isfile(candidate):
return candidate
except Exception:
pass
if os.path.exists(snapshots_dir): # Fallback: scan cache metadata explicitly (no downloads)
snapshots = os.listdir(snapshots_dir) cache = scan_cache_dir()
for repo in cache.repos:
if snapshots: if repo.repo_id == model_name:
return os.path.join(snapshots_dir, snapshots[0], gguf_in_repo) for revision in repo.revisions:
candidate = os.path.join(revision.snapshot_path, gguf_in_repo)
if os.path.isfile(candidate):
return candidate
return None return None
def main(): def main():
"""
Main function to find and print the model path.
"""
parser = argparse.ArgumentParser( parser = argparse.ArgumentParser(
description="Find the full GGUF path from the Hugging Face cache." description="Find the full GGUF path from the Hugging Face cache."
) )
parser.add_argument( parser.add_argument(
"model", type=str, help="The model name from Hugging Face" "model",
type=str,
help="Hugging Face model repo id (e.g. TheBloke/Mistral-7B-GGUF)",
) )
parser.add_argument( parser.add_argument(
"path", "path",
type=str, type=str,
help="The path to the GGUF file within the model repository", help="Relative path to the GGUF file inside the repo",
) )
args = parser.parse_args() args = parser.parse_args()
model_path = find_model_path(args.model, args.path) model_path = find_model_path(args.model, args.path)
if model_path is None: if model_path is None:
print( raise SystemExit("GGUF file not found in Hugging Face cache")
f"Error: Cached model not found. Model='{args.model}', GGUF='{args.path}', Cache dir='{CACHE_DIR}'",
file=sys.stderr,
)
sys.exit(1)
print(model_path, end="") print(model_path, end="")
+2
View File
@@ -1,3 +1,5 @@
runpod runpod
python-dotenv python-dotenv
openai openai
huggingface_hub
hf-transfer
+1 -7
View File
@@ -17,13 +17,7 @@ cleanup() {
CACHED_LLAMA_ARGS="" CACHED_LLAMA_ARGS=""
find_cached_path() { find_cached_path() {
local model_path CACHED_LLAMA_ARGS="-m $(python ./find_cached.py $LLAMA_CACHED_MODEL $LLAMA_CACHED_GGUF_PATH)"
model_path=$(python ./find_cached.py "$LLAMA_CACHED_MODEL" "$LLAMA_CACHED_GGUF_PATH")
if [ $? -ne 0 ] || [ -z "$model_path" ]; then
echo "start.sh: Error: Could not resolve cached model path. Check that LLAMA_CACHED_MODEL and LLAMA_CACHED_GGUF_PATH are correct and the model is fully cached on the network volume."
exit 1
fi
CACHED_LLAMA_ARGS="-m $model_path"
} }
# check if $LLAMA_CACHED_MODEL is set and not empty # check if $LLAMA_CACHED_MODEL is set and not empty