Merge pull request #3 from seurimas/master

Borrow vLLM concurrenccy logic a little.
This commit is contained in:
SvenBrnn
2025-09-14 09:48:58 +02:00
committed by GitHub
2 changed files with 31 additions and 7 deletions
+20
View File
@@ -20,6 +20,26 @@
"default": "phi3", "default": "phi3",
"advanced": false "advanced": false
} }
},
{
"key": "MAX_CONCURRENCY",
"input": {
"name": "Max Concurrency",
"type": "number",
"description": "Maximum number of concurrent requests to handle (default: 8)",
"default": 8,
"advanced": true
}
},
{
"key": "OLLAMA_NUM_PARALLEL",
"input": {
"name": "Parallel Requests",
"type": "string",
"description": "Maximum number of concurrent requests to handle (default: 4 or 1)",
"default": "",
"advanced": true
}
} }
] ]
} }
+4
View File
@@ -1,7 +1,10 @@
import runpod import runpod
import os
from utils import JobInput from utils import JobInput
from engine import OllamaEngine, OllamaOpenAiEngine from engine import OllamaEngine, OllamaOpenAiEngine
DEFAULT_MAX_CONCURRENCY = 8
max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
async def handler(job: any): async def handler(job: any):
# Just dump the whole input to the console and then return an {"ok": True} response # Just dump the whole input to the console and then return an {"ok": True} response
@@ -27,6 +30,7 @@ async def handler(job: any):
runpod.serverless.start( runpod.serverless.start(
{ {
"handler": handler, "handler": handler,
"concurrency_modifier": lambda x: max_concurrency,
"return_aggregate_stream": True, "return_aggregate_stream": True,
} }
) )