Borrow vLLM concurrenccy logic a little.

This commit is contained in:
Nicholas
2025-09-11 00:31:09 -05:00
parent 0df1db40e3
commit 51d38adcf3
2 changed files with 31 additions and 7 deletions
+4
View File
@@ -1,7 +1,10 @@
import runpod
import os
from utils import JobInput
from engine import OllamaEngine, OllamaOpenAiEngine
DEFAULT_MAX_CONCURRENCY = 8
max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
async def handler(job: any):
# Just dump the whole input to the console and then return an {"ok": True} response
@@ -27,6 +30,7 @@ async def handler(job: any):
runpod.serverless.start(
{
"handler": handler,
"concurrency_modifier": lambda x: max_concurrency,
"return_aggregate_stream": True,
}
)