Files
inference-worker/.runpod/hub.json
T

57 lines
2.3 KiB
JSON

{
"title": "llama.cpp",
"description": "Run llama.cpp inference using serverless RunPod workers!",
"type": "serverless",
"category": "language",
"iconUrl": "https://raw.githubusercontent.com/ggml-org/llama.cpp/master/media/llama1-icon-transparent.png",
"config": {
"runsOn": "GPU",
"gpuCount": 1,
"gpuIds": "AMPERE_16,AMPERE_24,ADA_24",
"containerDiskInGb": 32,
"presets": [],
"env": [
{
"key": "LLAMA_SERVER_CMD_ARGS",
"input": {
"name": "Command line arguments for llama-server",
"type": "string",
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port. If using caching, do not define -hf or -m here.",
"default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 999",
"advanced": false
}
},
{
"key": "LLAMA_CACHED_MODEL",
"input": {
"name": "Hugging Face Hub model name for cached model",
"type": "string",
"description": "Hugging Face Hub model name to use for the cached GGUF model. Leave empty to disable caching. Example: user/model-name",
"default": "",
"advanced": true
}
},
{
"key": "LLAMA_CACHED_GGUF_PATH",
"input": {
"name": "Path to GGUF file in the Hugging Face Hub model repository",
"type": "string",
"description": "Path to the GGUF file in the Hugging Face Hub model repository to use for caching. Example: model.gguf",
"default": "",
"advanced": true
}
},
{
"key": "MAX_CONCURRENCY",
"input": {
"name": "Maximum Concurrency",
"type": "number",
"description": "Maximum number of concurrent requests to handle (default: 8).",
"default": 8,
"advanced": true
}
}
]
}
}