Borrow vLLM concurrenccy logic a little.

This commit is contained in:
Nicholas
2025-09-11 00:31:09 -05:00
parent 0df1db40e3
commit 51d38adcf3
2 changed files with 31 additions and 7 deletions
+27 -7
View File
@@ -13,13 +13,33 @@
"env": [ "env": [
{ {
"key": "MODEL_NAME", "key": "MODEL_NAME",
"input": { "input": {
"name": "Model Name", "name": "Model Name",
"type": "string", "type": "string",
"description": "Name of a model to preload", "description": "Name of a model to preload",
"default": "phi3", "default": "phi3",
"advanced": false "advanced": false
} }
},
{
"key": "MAX_CONCURRENCY",
"input": {
"name": "Max Concurrency",
"type": "number",
"description": "Maximum number of concurrent requests to handle (default: 8)",
"default": 8,
"advanced": true
}
},
{
"key": "OLLAMA_NUM_PARALLEL",
"input": {
"name": "Parallel Requests",
"type": "string",
"description": "Maximum number of concurrent requests to handle (default: 4 or 1)",
"default": "",
"advanced": true
}
} }
] ]
} }
+4
View File
@@ -1,7 +1,10 @@
import runpod import runpod
import os
from utils import JobInput from utils import JobInput
from engine import OllamaEngine, OllamaOpenAiEngine from engine import OllamaEngine, OllamaOpenAiEngine
DEFAULT_MAX_CONCURRENCY = 8
max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
async def handler(job: any): async def handler(job: any):
# Just dump the whole input to the console and then return an {"ok": True} response # Just dump the whole input to the console and then return an {"ok": True} response
@@ -27,6 +30,7 @@ async def handler(job: any):
runpod.serverless.start( runpod.serverless.start(
{ {
"handler": handler, "handler": handler,
"concurrency_modifier": lambda x: max_concurrency,
"return_aggregate_stream": True, "return_aggregate_stream": True,
} }
) )