Final Bug fixes, configurable oai response role, served model name override, documentation
This commit is contained in:
@@ -111,6 +111,8 @@ This table provides a quick reference to the image tags you should use based on
|
|||||||
The way this works is that the first request will have a batch size of `DEFAULT_MIN_BATCH_SIZE`, and each subsequent request will have a batch size of `previous_batch_size * DEFAULT_BATCH_SIZE_GROWTH_FACTOR`. This will continue until the batch size reaches `DEFAULT_BATCH_SIZE`. E.g. for the default values, the batch sizes will be `1, 3, 9, 27, 50, 50, 50, ...`. You can also specify this per request, with inputs `max_batch_size`, `min_batch_size`, and `batch_size_growth_factor`. This has nothing to do with vLLM's internal batching, but rather the number of tokens sent in each HTTP request from the worker |
|
The way this works is that the first request will have a batch size of `DEFAULT_MIN_BATCH_SIZE`, and each subsequent request will have a batch size of `previous_batch_size * DEFAULT_BATCH_SIZE_GROWTH_FACTOR`. This will continue until the batch size reaches `DEFAULT_BATCH_SIZE`. E.g. for the default values, the batch sizes will be `1, 3, 9, 27, 50, 50, 50, ...`. You can also specify this per request, with inputs `max_batch_size`, `min_batch_size`, and `batch_size_growth_factor`. This has nothing to do with vLLM's internal batching, but rather the number of tokens sent in each HTTP request from the worker |
|
||||||
**OpenAI Settings**
|
**OpenAI Settings**
|
||||||
| `RAW_OPENAI_OUTPUT` | `1` | boolean as `int` |Enables raw OpenAI SSE format string output when streaming. **Required** to be enabled (which it is by default) for OpenAI compatibility. |
|
| `RAW_OPENAI_OUTPUT` | `1` | boolean as `int` |Enables raw OpenAI SSE format string output when streaming. **Required** to be enabled (which it is by default) for OpenAI compatibility. |
|
||||||
|
| `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | `None` | `str` |Overrides the name of the served model from model repo/path to specified name, which you will then be able to use the value for the `model` parameter when making OpenAI requests |
|
||||||
|
| `OPENAI_RESPONSE_ROLE` | `assistant` | `str` |Role of the LLM's Response in OpenAI Chat Completions. |
|
||||||
**Serverless Settings**
|
**Serverless Settings**
|
||||||
| `MAX_CONCURRENCY` | `300` | `int` |Max concurrent requests per worker. vLLM has an internal queue, so you don't have to worry about limiting by VRAM, this is for improving scaling/load balancing efficiency |
|
| `MAX_CONCURRENCY` | `300` | `int` |Max concurrent requests per worker. vLLM has an internal queue, so you don't have to worry about limiting by VRAM, this is for improving scaling/load balancing efficiency |
|
||||||
| `DISABLE_LOG_STATS` | `1` | boolean as `int` |Enables or disables vLLM stats logging. |
|
| `DISABLE_LOG_STATS` | `1` | boolean as `int` |Enables or disables vLLM stats logging. |
|
||||||
|
|||||||
+5
-2
@@ -112,6 +112,8 @@ class OpenAIvLLMEngine:
|
|||||||
def __init__(self, vllm_engine):
|
def __init__(self, vllm_engine):
|
||||||
self.config = vllm_engine.config
|
self.config = vllm_engine.config
|
||||||
self.llm = vllm_engine.llm
|
self.llm = vllm_engine.llm
|
||||||
|
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.config["model"]
|
||||||
|
self.response_role = os.getenv("OPENAI_RESPONSE_ROLE") or "assistant"
|
||||||
self.tokenizer = vllm_engine.tokenizer
|
self.tokenizer = vllm_engine.tokenizer
|
||||||
self.default_batch_size = vllm_engine.default_batch_size
|
self.default_batch_size = vllm_engine.default_batch_size
|
||||||
self.batch_size_growth_factor, self.min_batch_size = vllm_engine.batch_size_growth_factor, vllm_engine.min_batch_size
|
self.batch_size_growth_factor, self.min_batch_size = vllm_engine.batch_size_growth_factor, vllm_engine.min_batch_size
|
||||||
@@ -120,9 +122,10 @@ class OpenAIvLLMEngine:
|
|||||||
|
|
||||||
def _initialize_engines(self):
|
def _initialize_engines(self):
|
||||||
self.chat_engine = OpenAIServingChat(
|
self.chat_engine = OpenAIServingChat(
|
||||||
self.llm, self.config["model"], "assistant", self.tokenizer.tokenizer.chat_template
|
self.llm, self.served_model_name, self.response_role,
|
||||||
|
chat_template=self.tokenizer.tokenizer.chat_template
|
||||||
)
|
)
|
||||||
self.completion_engine = OpenAIServingCompletion(self.llm, self.config["model"])
|
self.completion_engine = OpenAIServingCompletion(self.llm, self.served_model_name)
|
||||||
|
|
||||||
async def generate(self, openai_request: JobInput):
|
async def generate(self, openai_request: JobInput):
|
||||||
if openai_request.openai_route == "/v1/models":
|
if openai_request.openai_route == "/v1/models":
|
||||||
|
|||||||
Reference in New Issue
Block a user