From 51d38adcf3a90d86d70afa2d90e75d9728776db1 Mon Sep 17 00:00:00 2001 From: Nicholas Date: Wed, 10 Sep 2025 22:05:01 -0500 Subject: [PATCH] Borrow vLLM concurrenccy logic a little. --- .runpod/hub.json | 34 +++++++++++++++++++++++++++------- src/handler.py | 4 ++++ 2 files changed, 31 insertions(+), 7 deletions(-) diff --git a/.runpod/hub.json b/.runpod/hub.json index 6bc2163..becd7ee 100644 --- a/.runpod/hub.json +++ b/.runpod/hub.json @@ -13,13 +13,33 @@ "env": [ { "key": "MODEL_NAME", - "input": { - "name": "Model Name", - "type": "string", - "description": "Name of a model to preload", - "default": "phi3", - "advanced": false - } + "input": { + "name": "Model Name", + "type": "string", + "description": "Name of a model to preload", + "default": "phi3", + "advanced": false + } + }, + { + "key": "MAX_CONCURRENCY", + "input": { + "name": "Max Concurrency", + "type": "number", + "description": "Maximum number of concurrent requests to handle (default: 8)", + "default": 8, + "advanced": true + } + }, + { + "key": "OLLAMA_NUM_PARALLEL", + "input": { + "name": "Parallel Requests", + "type": "string", + "description": "Maximum number of concurrent requests to handle (default: 4 or 1)", + "default": "", + "advanced": true + } } ] } diff --git a/src/handler.py b/src/handler.py index d0bcfd7..26354fe 100644 --- a/src/handler.py +++ b/src/handler.py @@ -1,7 +1,10 @@ import runpod +import os from utils import JobInput from engine import OllamaEngine, OllamaOpenAiEngine +DEFAULT_MAX_CONCURRENCY = 8 +max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY)) async def handler(job: any): # Just dump the whole input to the console and then return an {"ok": True} response @@ -27,6 +30,7 @@ async def handler(job: any): runpod.serverless.start( { "handler": handler, + "concurrency_modifier": lambda x: max_concurrency, "return_aggregate_stream": True, } ) \ No newline at end of file