4 Commits
3 changed files with 47 additions and 24 deletions
+35 -23
View File
@@ -1,57 +1,69 @@
"""
Finds the full LLM GGUF path from the Hugging Face cache.
Finds the full GGUF path from the Hugging Face cache using huggingface_hub.
"""
import os
import argparse
CACHE_DIR = "/runpod-volume/huggingface-cache/hub"
import os
from huggingface_hub import snapshot_download, scan_cache_dir
def find_model_path(model_name, gguf_in_repo="model.gguf"):
def find_model_path(model_name: str, gguf_in_repo: str) -> str | None:
"""
Find the path to a cached model.
Resolve the GGUF file path from the Hugging Face cache.
Args:
model_name: The model name from Hugging Face
model_name: Hugging Face repo id (e.g. TheBloke/Mistral-7B-GGUF)
gguf_in_repo: Relative path to the GGUF file inside the repo
Returns:
The full path to the cached model, or None if not found
Full filesystem path to the GGUF file, or None if not found
"""
cache_name = model_name.replace("/", "--")
snapshots_dir = os.path.join(
CACHE_DIR, f"models--{cache_name}", "snapshots"
)
# First try the official way: resolve snapshot path from cache
try:
snapshot_path = snapshot_download(
repo_id=model_name,
local_files_only=True
)
candidate = os.path.join(snapshot_path, gguf_in_repo)
if os.path.isfile(candidate):
return candidate
except Exception:
pass
if os.path.exists(snapshots_dir):
snapshots = os.listdir(snapshots_dir)
if snapshots:
return os.path.join(snapshots_dir, snapshots[0], gguf_in_repo)
# Fallback: scan cache metadata explicitly (no downloads)
cache = scan_cache_dir()
for repo in cache.repos:
if repo.repo_id == model_name:
for revision in repo.revisions:
candidate = os.path.join(revision.snapshot_path, gguf_in_repo)
if os.path.isfile(candidate):
return candidate
return None
def main():
"""
Main function to find and print the model path.
"""
parser = argparse.ArgumentParser(
description="Find the full GGUF path from the Hugging Face cache."
)
parser.add_argument(
"model", type=str, help="The model name from Hugging Face"
"model",
type=str,
help="Hugging Face model repo id (e.g. TheBloke/Mistral-7B-GGUF)",
)
parser.add_argument(
"path",
type=str,
help="The path to the GGUF file within the model repository",
help="Relative path to the GGUF file inside the repo",
)
args = parser.parse_args()
model_path = find_model_path(args.model, args.path)
if model_path is None:
raise SystemExit("GGUF file not found in Hugging Face cache")
print(model_path, end="")
+2
View File
@@ -1,3 +1,5 @@
runpod
python-dotenv
openai
huggingface_hub
hf-transfer
+10 -1
View File
@@ -56,7 +56,7 @@ echo "start.sh: Stopping existing llama-server instances (if any)..."
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
# it contains a.e. "-hf modelname --ctx-size 4096 -ngl 999".
echo "start.sh: Running llama-server $LLAMA_SERVER_CMD_ARGS --port 3098"
echo "start.sh: Running /app/llama-server $CACHED_LLAMA_ARGS $LLAMA_SERVER_CMD_ARGS --port 3098"
touch llama.server.log
@@ -65,6 +65,8 @@ LD_LIBRARY_PATH=/app /app/llama-server $CACHED_LLAMA_ARGS $LLAMA_SERVER_CMD_ARGS
LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command
tries_so_far=0
check_server_is_running() {
echo "start.sh: Checking if llama-server is done initializing..."
@@ -74,6 +76,13 @@ check_server_is_running() {
return 1 # failure
fi
tries_so_far=$((tries_so_far + 1))
if [ $tries_so_far -ge 120 ]; then
echo "start.sh: Error: llama-server did not start within 60 seconds."
exit 1
fi
# check if the process is still running
if ! kill -0 $LLAMA_SERVER_PID 2>/dev/null; then
echo "start.sh: Error: llama-server process has exited unexpectedly."