Merge pull request #3 from seurimas/master
Borrow vLLM concurrenccy logic a little.
This commit is contained in:
@@ -20,6 +20,26 @@
|
|||||||
"default": "phi3",
|
"default": "phi3",
|
||||||
"advanced": false
|
"advanced": false
|
||||||
}
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"key": "MAX_CONCURRENCY",
|
||||||
|
"input": {
|
||||||
|
"name": "Max Concurrency",
|
||||||
|
"type": "number",
|
||||||
|
"description": "Maximum number of concurrent requests to handle (default: 8)",
|
||||||
|
"default": 8,
|
||||||
|
"advanced": true
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"key": "OLLAMA_NUM_PARALLEL",
|
||||||
|
"input": {
|
||||||
|
"name": "Parallel Requests",
|
||||||
|
"type": "string",
|
||||||
|
"description": "Maximum number of concurrent requests to handle (default: 4 or 1)",
|
||||||
|
"default": "",
|
||||||
|
"advanced": true
|
||||||
|
}
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,7 +1,10 @@
|
|||||||
import runpod
|
import runpod
|
||||||
|
import os
|
||||||
from utils import JobInput
|
from utils import JobInput
|
||||||
from engine import OllamaEngine, OllamaOpenAiEngine
|
from engine import OllamaEngine, OllamaOpenAiEngine
|
||||||
|
|
||||||
|
DEFAULT_MAX_CONCURRENCY = 8
|
||||||
|
max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY))
|
||||||
|
|
||||||
async def handler(job: any):
|
async def handler(job: any):
|
||||||
# Just dump the whole input to the console and then return an {"ok": True} response
|
# Just dump the whole input to the console and then return an {"ok": True} response
|
||||||
@@ -27,6 +30,7 @@ async def handler(job: any):
|
|||||||
runpod.serverless.start(
|
runpod.serverless.start(
|
||||||
{
|
{
|
||||||
"handler": handler,
|
"handler": handler,
|
||||||
|
"concurrency_modifier": lambda x: max_concurrency,
|
||||||
"return_aggregate_stream": True,
|
"return_aggregate_stream": True,
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
Reference in New Issue
Block a user