requested changes
This commit is contained in:
+35
-6
@@ -4,7 +4,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -34,11 +34,11 @@ For complete configuration options, see the [full configuration documentation](h
|
|||||||
|
|
||||||
## API Usage
|
## API Usage
|
||||||
|
|
||||||
This worker supports two API formats: **Runpod native** and **OpenAI-compatible**.
|
This worker supports two API formats: **RunPod native** and **OpenAI-compatible**.
|
||||||
|
|
||||||
### Runpod Native API
|
### RunPod Native API
|
||||||
|
|
||||||
For testing directly in the Runpod UI, use these examples in your endpoint's request tab.
|
For testing directly in the RunPod UI, use these examples in your endpoint's request tab.
|
||||||
|
|
||||||
#### Chat Completions
|
#### Chat Completions
|
||||||
|
|
||||||
@@ -104,7 +104,7 @@ For direct text generation without chat format:
|
|||||||
|
|
||||||
### OpenAI-Compatible API
|
### OpenAI-Compatible API
|
||||||
|
|
||||||
For external clients and SDKs, use the `/openai/v1` path prefix with your Runpod API key.
|
For external clients and SDKs, use the `/openai/v1` path prefix with your RunPod API key.
|
||||||
|
|
||||||
#### Chat Completions
|
#### Chat Completions
|
||||||
|
|
||||||
@@ -157,6 +157,35 @@ For external clients and SDKs, use the `/openai/v1` path prefix with your Runpod
|
|||||||
{}
|
{}
|
||||||
```
|
```
|
||||||
|
|
||||||
|
#### OpenAI Responses API
|
||||||
|
|
||||||
|
**Path:** `/openai/v1/responses`
|
||||||
|
|
||||||
|
Supports the [OpenAI Responses API](https://platform.openai.com/docs/api-reference/responses) format. Note: this route bypasses the RunPod queue and is served directly — use `/openai/` prefixed paths rather than the RunPod job queue for these endpoints.
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"model": "meta-llama/Llama-3.1-8B-Instruct",
|
||||||
|
"input": "Tell me a joke."
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
#### Anthropic Messages API
|
||||||
|
|
||||||
|
**Path:** `/anthropic/v1/messages`
|
||||||
|
|
||||||
|
Supports the [Anthropic Messages API](https://docs.anthropic.com/en/api/messages) format. Served directly, bypassing the RunPod queue.
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"model": "meta-llama/Llama-3.1-8B-Instruct",
|
||||||
|
"max_tokens": 256,
|
||||||
|
"messages": [
|
||||||
|
{"role": "user", "content": "Hello!"}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
#### Response Format
|
#### Response Format
|
||||||
|
|
||||||
Both APIs return the same response format:
|
Both APIs return the same response format:
|
||||||
@@ -180,7 +209,7 @@ Both APIs return the same response format:
|
|||||||
|
|
||||||
Below are minimal `python` snippets so you can copy-paste to get started quickly.
|
Below are minimal `python` snippets so you can copy-paste to get started quickly.
|
||||||
|
|
||||||
> Replace `<ENDPOINT_ID>` with your endpoint ID and `<API_KEY>` with a [Runpod API key](https://docs.runpod.io/get-started/api-keys).
|
> Replace `<ENDPOINT_ID>` with your endpoint ID and `<API_KEY>` with a [RunPod API key](https://docs.runpod.io/get-started/api-keys).
|
||||||
|
|
||||||
### OpenAI compatible API
|
### OpenAI compatible API
|
||||||
|
|
||||||
|
|||||||
+1
-6
@@ -2,7 +2,7 @@ FROM nvidia/cuda:12.9.1-base-ubuntu22.04
|
|||||||
|
|
||||||
RUN apt-get update -y \
|
RUN apt-get update -y \
|
||||||
&& apt-get install -y python3-pip curl \
|
&& apt-get install -y python3-pip curl \
|
||||||
&& curl -LsSf https://astral.sh/uv/install.sh | sh
|
&& curl -LsSf https://astral.sh/uv/0.10.9/install.sh | sh
|
||||||
|
|
||||||
ENV PATH="/root/.local/bin:$PATH"
|
ENV PATH="/root/.local/bin:$PATH"
|
||||||
|
|
||||||
@@ -25,7 +25,6 @@ ARG QUANTIZATION=""
|
|||||||
ARG MODEL_REVISION=""
|
ARG MODEL_REVISION=""
|
||||||
ARG TOKENIZER_REVISION=""
|
ARG TOKENIZER_REVISION=""
|
||||||
ARG VLLM_NIGHTLY="false"
|
ARG VLLM_NIGHTLY="false"
|
||||||
ARG LMCACHE="false"
|
|
||||||
|
|
||||||
ENV MODEL_NAME=$MODEL_NAME \
|
ENV MODEL_NAME=$MODEL_NAME \
|
||||||
MODEL_REVISION=$MODEL_REVISION \
|
MODEL_REVISION=$MODEL_REVISION \
|
||||||
@@ -47,10 +46,6 @@ ENV MODEL_NAME=$MODEL_NAME \
|
|||||||
|
|
||||||
ENV PYTHONPATH="/:/vllm-workspace"
|
ENV PYTHONPATH="/:/vllm-workspace"
|
||||||
|
|
||||||
RUN if [ "${LMCACHE}" = "true" ]; then \
|
|
||||||
uv pip install --system lmcache; \
|
|
||||||
fi
|
|
||||||
|
|
||||||
RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \
|
RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \
|
||||||
uv pip install --system -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \
|
uv pip install --system -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \
|
||||||
apt-get update && apt-get install -y git && rm -rf /var/lib/apt/lists/* && \
|
apt-get update && apt-get install -y git && rm -rf /var/lib/apt/lists/* && \
|
||||||
|
|||||||
@@ -3,12 +3,13 @@ pandas
|
|||||||
pyarrow
|
pyarrow
|
||||||
runpod
|
runpod
|
||||||
huggingface-hub
|
huggingface-hub
|
||||||
|
lmcache==0.4.2
|
||||||
packaging>=24.2
|
packaging>=24.2
|
||||||
typing-extensions>=4.8.0
|
typing-extensions>=4.8.0
|
||||||
pydantic
|
pydantic
|
||||||
pydantic-settings
|
pydantic-settings
|
||||||
hf-transfer
|
hf-transfer
|
||||||
transformers>=5.2.0
|
transformers>=5.2.0,<5.3.0
|
||||||
bitsandbytes>=0.45.0
|
bitsandbytes>=0.45.0
|
||||||
kernels
|
kernels
|
||||||
torch-c-dlpack-ext
|
torch-c-dlpack-ext
|
||||||
|
|||||||
+101
-12
@@ -384,36 +384,112 @@ class OpenAIvLLMEngine(vLLMEngine):
|
|||||||
yield batch
|
yield batch
|
||||||
|
|
||||||
async def _handle_responses_request(self, openai_request: JobInput):
|
async def _handle_responses_request(self, openai_request: JobInput):
|
||||||
|
request_id = getattr(openai_request, "request_id", "unknown")
|
||||||
|
|
||||||
try:
|
try:
|
||||||
request = ResponsesRequest(**openai_request.openai_input)
|
request = ResponsesRequest(**openai_request.openai_input)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
yield create_error_response(str(e)).model_dump()
|
logging.error(
|
||||||
|
"Invalid ResponsesRequest JSON: %s",
|
||||||
|
e,
|
||||||
|
extra={"request_id": request_id}
|
||||||
|
)
|
||||||
|
yield create_error_response(
|
||||||
|
"Invalid request format",
|
||||||
|
err_type="BadRequestError"
|
||||||
|
).model_dump()
|
||||||
return
|
return
|
||||||
|
|
||||||
dummy_request = DummyRequest()
|
dummy_request = DummyRequest()
|
||||||
response = await self.responses_engine.create_responses(request, raw_request=dummy_request)
|
try:
|
||||||
|
response = await self.responses_engine.create_responses(request, raw_request=dummy_request)
|
||||||
|
except Exception as e:
|
||||||
|
logging.error(
|
||||||
|
"Failed to create Responses: %s",
|
||||||
|
e,
|
||||||
|
extra={"request_id": request_id},
|
||||||
|
exc_info=True
|
||||||
|
)
|
||||||
|
if isinstance(response, ErrorResponse):
|
||||||
|
yield response.model_dump()
|
||||||
|
else:
|
||||||
|
yield create_error_response(
|
||||||
|
"Internal server error during response generation",
|
||||||
|
err_type="InternalServerError"
|
||||||
|
).model_dump()
|
||||||
|
return
|
||||||
|
|
||||||
if isinstance(response, (ErrorResponse, ResponsesResponse)):
|
if isinstance(response, (ErrorResponse, ResponsesResponse)):
|
||||||
yield response.model_dump()
|
yield response.model_dump()
|
||||||
return
|
return
|
||||||
|
|
||||||
async for event in response:
|
try:
|
||||||
event_type = getattr(event, "type", "unknown")
|
async for event in response:
|
||||||
yield f"event: {event_type}\ndata: {event.model_dump_json(indent=None)}\n\n"
|
if not hasattr(event, "type"):
|
||||||
|
continue
|
||||||
|
event_type = getattr(event, "type", "unknown")
|
||||||
|
yield f"event: {event_type}\ndata: {event.model_dump_json(indent=None)}\n\n"
|
||||||
|
except Exception as e:
|
||||||
|
logging.error(
|
||||||
|
"Error processing responses stream: %s",
|
||||||
|
e,
|
||||||
|
extra={"request_id": request_id},
|
||||||
|
exc_info=True
|
||||||
|
)
|
||||||
|
yield create_error_response(
|
||||||
|
"Streaming response failed",
|
||||||
|
err_type="InternalServerError"
|
||||||
|
).model_dump()
|
||||||
async def _handle_messages_request(self, openai_request: JobInput):
|
async def _handle_messages_request(self, openai_request: JobInput):
|
||||||
|
request_id = getattr(openai_request, "request_id", "unknown")
|
||||||
|
|
||||||
try:
|
try:
|
||||||
request = AnthropicMessagesRequest(**openai_request.openai_input)
|
request = AnthropicMessagesRequest(**openai_request.openai_input)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
yield create_error_response(str(e)).model_dump()
|
logging.error(
|
||||||
|
"Invalid AnthropicMessagesRequest: %s",
|
||||||
|
e,
|
||||||
|
extra={"request_id": request_id}
|
||||||
|
)
|
||||||
|
yield AnthropicErrorResponse(
|
||||||
|
error=AnthropicError(
|
||||||
|
type="invalid_request_error",
|
||||||
|
message="Invalid request format"
|
||||||
|
)
|
||||||
|
).model_dump()
|
||||||
return
|
return
|
||||||
|
|
||||||
dummy_request = DummyRequest()
|
dummy_request = DummyRequest()
|
||||||
response = await self.messages_engine.create_messages(request, raw_request=dummy_request)
|
|
||||||
|
try:
|
||||||
|
response = await self.messages_engine.create_messages(request, raw_request=dummy_request)
|
||||||
|
except Exception as e:
|
||||||
|
logging.error(
|
||||||
|
"Failed to create messages: %s",
|
||||||
|
e,
|
||||||
|
extra={"request_id": request_id},
|
||||||
|
exc_info=True
|
||||||
|
)
|
||||||
|
if isinstance(response, ErrorResponse):
|
||||||
|
error_type = getattr(response, "type", "internal_error")
|
||||||
|
error_message = getattr(response, "message", str(e)[:200])
|
||||||
|
yield AnthropicErrorResponse(
|
||||||
|
error=AnthropicError(type=error_type, message=error_message)
|
||||||
|
).model_dump()
|
||||||
|
else:
|
||||||
|
yield AnthropicErrorResponse(
|
||||||
|
error=AnthropicError(
|
||||||
|
type="internal_error",
|
||||||
|
message="Failed to generate messages"
|
||||||
|
)
|
||||||
|
).model_dump()
|
||||||
|
return
|
||||||
|
|
||||||
if isinstance(response, ErrorResponse):
|
if isinstance(response, ErrorResponse):
|
||||||
|
error_type = getattr(response, "type", "internal_error")
|
||||||
|
error_message = getattr(response, "message", "Unknown error")
|
||||||
yield AnthropicErrorResponse(
|
yield AnthropicErrorResponse(
|
||||||
error=AnthropicError(type=response.error.type, message=response.error.message)
|
error=AnthropicError(type=error_type, message=error_message)
|
||||||
).model_dump()
|
).model_dump()
|
||||||
return
|
return
|
||||||
|
|
||||||
@@ -421,6 +497,19 @@ class OpenAIvLLMEngine(vLLMEngine):
|
|||||||
yield response.model_dump(exclude_none=True)
|
yield response.model_dump(exclude_none=True)
|
||||||
return
|
return
|
||||||
|
|
||||||
async for chunk in response:
|
try:
|
||||||
yield chunk
|
async for chunk in response:
|
||||||
|
yield chunk
|
||||||
|
except Exception as e:
|
||||||
|
logging.error(
|
||||||
|
"Error streaming messages: %s",
|
||||||
|
e,
|
||||||
|
extra={"request_id": request_id},
|
||||||
|
exc_info=True
|
||||||
|
)
|
||||||
|
yield AnthropicErrorResponse(
|
||||||
|
error=AnthropicError(
|
||||||
|
type="internal_error",
|
||||||
|
message="Error while streaming messages"
|
||||||
|
)
|
||||||
|
).model_dump()
|
||||||
|
|||||||
+26
-9
@@ -402,15 +402,32 @@ def get_engine_args():
|
|||||||
logging.warning("Overriding MAX_PARALLEL_LOADING_WORKERS with None because more than 1 GPU is available.")
|
logging.warning("Overriding MAX_PARALLEL_LOADING_WORKERS with None because more than 1 GPU is available.")
|
||||||
|
|
||||||
# LMCache requires HMA to be disabled
|
# LMCache requires HMA to be disabled
|
||||||
_kv_transfer = args.get("kv_transfer_config")
|
try:
|
||||||
_kv_offload = args.get("kv_offloading_backend")
|
_kv_transfer = args.get("kv_transfer_config")
|
||||||
_lmcache_active = _kv_offload == "lmcache" or (
|
_kv_offload = args.get("kv_offloading_backend")
|
||||||
isinstance(_kv_transfer, dict)
|
|
||||||
and "lmcache" in str(_kv_transfer.get("kv_connector", "")).lower()
|
lmcache_detected = _kv_offload == "lmcache" or (
|
||||||
)
|
isinstance(_kv_transfer, dict)
|
||||||
if _lmcache_active and not args.get("disable_hybrid_kv_cache_manager"):
|
and isinstance(_kv_transfer.get("kv_connector"), str)
|
||||||
args["disable_hybrid_kv_cache_manager"] = True
|
and "lmcache" in _kv_transfer.get("kv_connector", "").lower()
|
||||||
logging.info("LMCache detected: automatically setting disable_hybrid_kv_cache_manager=True")
|
)
|
||||||
|
|
||||||
|
if lmcache_detected and not args.get("disable_hybrid_kv_cache_manager"):
|
||||||
|
args["disable_hybrid_kv_cache_manager"] = True
|
||||||
|
args["kv_offloading_backend"] = None
|
||||||
|
args["kv_transfer_config"] = None
|
||||||
|
logging.info("LMCache detected: automatically setting disable_hybrid_kv_cache_manager=True and clearing conflicting settings")
|
||||||
|
elif lmcache_detected and args.get("disable_hybrid_kv_cache_manager") is False:
|
||||||
|
logging.warning(
|
||||||
|
"LMCache configuration detected but disabled: "
|
||||||
|
"disable_hybrid_kv_cache_manager must be False when using LMCache"
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
logging.error(
|
||||||
|
"Failed to check LMCache configuration: %s",
|
||||||
|
e,
|
||||||
|
exc_info=True
|
||||||
|
)
|
||||||
|
|
||||||
# Deprecated env args backwards compatibility
|
# Deprecated env args backwards compatibility
|
||||||
if args.get("kv_cache_dtype") == "fp8_e5m2":
|
if args.get("kv_cache_dtype") == "fp8_e5m2":
|
||||||
|
|||||||
Reference in New Issue
Block a user