Borrow vLLM concurrenccy logic a little.
This commit is contained in:
+27
-7
@@ -13,13 +13,33 @@
|
||||
"env": [
|
||||
{
|
||||
"key": "MODEL_NAME",
|
||||
"input": {
|
||||
"name": "Model Name",
|
||||
"type": "string",
|
||||
"description": "Name of a model to preload",
|
||||
"default": "phi3",
|
||||
"advanced": false
|
||||
}
|
||||
"input": {
|
||||
"name": "Model Name",
|
||||
"type": "string",
|
||||
"description": "Name of a model to preload",
|
||||
"default": "phi3",
|
||||
"advanced": false
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "MAX_CONCURRENCY",
|
||||
"input": {
|
||||
"name": "Max Concurrency",
|
||||
"type": "number",
|
||||
"description": "Maximum number of concurrent requests to handle (default: 8)",
|
||||
"default": 8,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "OLLAMA_NUM_PARALLEL",
|
||||
"input": {
|
||||
"name": "Parallel Requests",
|
||||
"type": "string",
|
||||
"description": "Maximum number of concurrent requests to handle (default: 4 or 1)",
|
||||
"default": "",
|
||||
"advanced": true
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
@@ -1,7 +1,10 @@
|
||||
import runpod
|
||||
import os
|
||||
from utils import JobInput
|
||||
from engine import OllamaEngine, OllamaOpenAiEngine
|
||||
|
||||
DEFAULT_MAX_CONCURRENCY = 8
|
||||
max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
|
||||
|
||||
async def handler(job: any):
|
||||
# Just dump the whole input to the console and then return an {"ok": True} response
|
||||
@@ -27,6 +30,7 @@ async def handler(job: any):
|
||||
runpod.serverless.start(
|
||||
{
|
||||
"handler": handler,
|
||||
"concurrency_modifier": lambda x: max_concurrency,
|
||||
"return_aggregate_stream": True,
|
||||
}
|
||||
)
|
||||
Reference in New Issue
Block a user