Final Bug fixes, configurable oai response role, served model name override, documentation

This commit is contained in:
alpayariyak
2024-02-23 04:14:03 +00:00
parent b42d45ce0f
commit 708f68d7f8
2 changed files with 7 additions and 2 deletions
+2
View File
@@ -111,6 +111,8 @@ This table provides a quick reference to the image tags you should use based on
The way this works is that the first request will have a batch size of `DEFAULT_MIN_BATCH_SIZE`, and each subsequent request will have a batch size of `previous_batch_size * DEFAULT_BATCH_SIZE_GROWTH_FACTOR`. This will continue until the batch size reaches `DEFAULT_BATCH_SIZE`. E.g. for the default values, the batch sizes will be `1, 3, 9, 27, 50, 50, 50, ...`. You can also specify this per request, with inputs `max_batch_size`, `min_batch_size`, and `batch_size_growth_factor`. This has nothing to do with vLLM's internal batching, but rather the number of tokens sent in each HTTP request from the worker | The way this works is that the first request will have a batch size of `DEFAULT_MIN_BATCH_SIZE`, and each subsequent request will have a batch size of `previous_batch_size * DEFAULT_BATCH_SIZE_GROWTH_FACTOR`. This will continue until the batch size reaches `DEFAULT_BATCH_SIZE`. E.g. for the default values, the batch sizes will be `1, 3, 9, 27, 50, 50, 50, ...`. You can also specify this per request, with inputs `max_batch_size`, `min_batch_size`, and `batch_size_growth_factor`. This has nothing to do with vLLM's internal batching, but rather the number of tokens sent in each HTTP request from the worker |
**OpenAI Settings** **OpenAI Settings**
| `RAW_OPENAI_OUTPUT` | `1` | boolean as `int` |Enables raw OpenAI SSE format string output when streaming. **Required** to be enabled (which it is by default) for OpenAI compatibility. | | `RAW_OPENAI_OUTPUT` | `1` | boolean as `int` |Enables raw OpenAI SSE format string output when streaming. **Required** to be enabled (which it is by default) for OpenAI compatibility. |
| `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | `None` | `str` |Overrides the name of the served model from model repo/path to specified name, which you will then be able to use the value for the `model` parameter when making OpenAI requests |
| `OPENAI_RESPONSE_ROLE` | `assistant` | `str` |Role of the LLM's Response in OpenAI Chat Completions. |
**Serverless Settings** **Serverless Settings**
| `MAX_CONCURRENCY` | `300` | `int` |Max concurrent requests per worker. vLLM has an internal queue, so you don't have to worry about limiting by VRAM, this is for improving scaling/load balancing efficiency | | `MAX_CONCURRENCY` | `300` | `int` |Max concurrent requests per worker. vLLM has an internal queue, so you don't have to worry about limiting by VRAM, this is for improving scaling/load balancing efficiency |
| `DISABLE_LOG_STATS` | `1` | boolean as `int` |Enables or disables vLLM stats logging. | | `DISABLE_LOG_STATS` | `1` | boolean as `int` |Enables or disables vLLM stats logging. |
+5 -2
View File
@@ -112,6 +112,8 @@ class OpenAIvLLMEngine:
def __init__(self, vllm_engine): def __init__(self, vllm_engine):
self.config = vllm_engine.config self.config = vllm_engine.config
self.llm = vllm_engine.llm self.llm = vllm_engine.llm
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.config["model"]
self.response_role = os.getenv("OPENAI_RESPONSE_ROLE") or "assistant"
self.tokenizer = vllm_engine.tokenizer self.tokenizer = vllm_engine.tokenizer
self.default_batch_size = vllm_engine.default_batch_size self.default_batch_size = vllm_engine.default_batch_size
self.batch_size_growth_factor, self.min_batch_size = vllm_engine.batch_size_growth_factor, vllm_engine.min_batch_size self.batch_size_growth_factor, self.min_batch_size = vllm_engine.batch_size_growth_factor, vllm_engine.min_batch_size
@@ -120,9 +122,10 @@ class OpenAIvLLMEngine:
def _initialize_engines(self): def _initialize_engines(self):
self.chat_engine = OpenAIServingChat( self.chat_engine = OpenAIServingChat(
self.llm, self.config["model"], "assistant", self.tokenizer.tokenizer.chat_template self.llm, self.served_model_name, self.response_role,
chat_template=self.tokenizer.tokenizer.chat_template
) )
self.completion_engine = OpenAIServingCompletion(self.llm, self.config["model"]) self.completion_engine = OpenAIServingCompletion(self.llm, self.served_model_name)
async def generate(self, openai_request: JobInput): async def generate(self, openai_request: JobInput):
if openai_request.openai_route == "/v1/models": if openai_request.openai_route == "/v1/models":