change project to work with llama.cpp instead of Ollama

This commit is contained in:
mags0ft
2025-11-15 14:03:10 +01:00
parent d58239e290
commit 3a8d2decfd
21 changed files with 622 additions and 497 deletions
+34 -44
View File
@@ -1,46 +1,36 @@
{
"title": "Runpod Worker Ollama",
"description": "A serverless Ollama Worker for Runpod",
"type": "serverless",
"category": "language",
"iconUrl": "https://ollama.com/public/ollama.png",
"config": {
"runsOn": "GPU",
"gpuCount": 1,
"gpuIds": "AMPERE_16,AMPERE_24,ADA_24",
"containerDiskInGb": 20,
"presets": [],
"env": [
{
"key": "OLLAMA_MODEL_NAME",
"input": {
"name": "Model Name",
"type": "string",
"description": "Name of a model to preload",
"default": "phi3",
"advanced": false
}
},
{
"key": "MAX_CONCURRENCY",
"input": {
"name": "Max Concurrency",
"type": "number",
"description": "Maximum number of concurrent requests to handle (default: 8)",
"default": 8,
"advanced": true
}
},
{
"key": "OLLAMA_NUM_PARALLEL",
"input": {
"name": "Parallel Requests",
"type": "string",
"description": "Maximum number of concurrent requests to handle (default: 4 or 1)",
"default": "",
"advanced": true
}
}
]
}
"title": "llama.cpp inference",
"description": "Run llama.cpp inference using serverless RunPod workers!",
"type": "serverless",
"category": "language",
"iconUrl": "https://raw.githubusercontent.com/ggml-org/llama.cpp/master/media/llama1-icon-transparent.png",
"config": {
"runsOn": "GPU",
"gpuCount": 1,
"gpuIds": "AMPERE_16,AMPERE_24,ADA_24",
"containerDiskInGb": 32,
"presets": [],
"env": [
{
"key": "LLAMA_SERVER_CMD_ARGS",
"input": {
"name": "Model Name",
"type": "string",
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.",
"default": "-hf unsloth/Mistral-Small-3.2-24B-Instruct-2506-GGUF:Q4_K_M -ctx_size 4096",
"advanced": false
}
},
{
"key": "MAX_CONCURRENCY",
"input": {
"name": "Maximum Concurrency",
"type": "number",
"description": "Maximum number of concurrent requests to handle (default: 8).",
"default": 8,
"advanced": true
}
}
]
}
}
+22 -22
View File
@@ -1,24 +1,24 @@
{
"tests": [
{
"name": "execute_a_prompt",
"input": {
"prompt": "Say: Hallo World!"
},
"timeout": 120000
}
],
"config": {
"gpuTypeId": "NVIDIA GeForce RTX 4090",
"gpuCount": 1,
"env": [
{
"key": "OLLAMA_MODEL_NAME",
"value": "phi3"
}
"tests": [
{
"name": "execute_a_prompt",
"input": {
"prompt": "Hi! Who are you?"
},
"timeout": 120000
}
],
"allowedCudaVersions": [
"12.8"
]
}
}
"config": {
"gpuTypeId": "NVIDIA GeForce RTX 4090",
"gpuCount": 1,
"env": [
{
"key": "LLAMA_SERVER_CMD_ARGS",
"value": "-hf unsloth/Mistral-Small-3.2-24B-Instruct-2506-GGUF:Q4_K_M -ctx_size 4096"
}
],
"allowedCudaVersions": [
"12.8"
]
}
}