add caching support for models in start.sh and update documentation
This commit is contained in:
+22
-2
@@ -16,11 +16,31 @@
|
||||
"input": {
|
||||
"name": "Command line arguments for llama-server",
|
||||
"type": "string",
|
||||
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.",
|
||||
"default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 99",
|
||||
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port. If using caching, do not define -hf or -m here.",
|
||||
"default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 999",
|
||||
"advanced": false
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "LLAMA_CACHED_MODEL",
|
||||
"input": {
|
||||
"name": "Hugging Face Hub model name for cached model",
|
||||
"type": "string",
|
||||
"description": "Hugging Face Hub model name to use for the cached GGUF model. Leave empty to disable caching. Example: user/model-name",
|
||||
"default": "",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "LLAMA_CACHED_GGUF_PATH",
|
||||
"input": {
|
||||
"name": "Path to GGUF file in the Hugging Face Hub model repository",
|
||||
"type": "string",
|
||||
"description": "Path to the GGUF file in the Hugging Face Hub model repository to use for caching. Example: model.gguf",
|
||||
"default": "",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "MAX_CONCURRENCY",
|
||||
"input": {
|
||||
|
||||
Reference in New Issue
Block a user