requested changes
This commit is contained in:
+35
-6
@@ -4,7 +4,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
|
||||
|
||||
---
|
||||
|
||||
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
||||
[](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
|
||||
|
||||
---
|
||||
|
||||
@@ -34,11 +34,11 @@ For complete configuration options, see the [full configuration documentation](h
|
||||
|
||||
## API Usage
|
||||
|
||||
This worker supports two API formats: **Runpod native** and **OpenAI-compatible**.
|
||||
This worker supports two API formats: **RunPod native** and **OpenAI-compatible**.
|
||||
|
||||
### Runpod Native API
|
||||
### RunPod Native API
|
||||
|
||||
For testing directly in the Runpod UI, use these examples in your endpoint's request tab.
|
||||
For testing directly in the RunPod UI, use these examples in your endpoint's request tab.
|
||||
|
||||
#### Chat Completions
|
||||
|
||||
@@ -104,7 +104,7 @@ For direct text generation without chat format:
|
||||
|
||||
### OpenAI-Compatible API
|
||||
|
||||
For external clients and SDKs, use the `/openai/v1` path prefix with your Runpod API key.
|
||||
For external clients and SDKs, use the `/openai/v1` path prefix with your RunPod API key.
|
||||
|
||||
#### Chat Completions
|
||||
|
||||
@@ -157,6 +157,35 @@ For external clients and SDKs, use the `/openai/v1` path prefix with your Runpod
|
||||
{}
|
||||
```
|
||||
|
||||
#### OpenAI Responses API
|
||||
|
||||
**Path:** `/openai/v1/responses`
|
||||
|
||||
Supports the [OpenAI Responses API](https://platform.openai.com/docs/api-reference/responses) format. Note: this route bypasses the RunPod queue and is served directly — use `/openai/` prefixed paths rather than the RunPod job queue for these endpoints.
|
||||
|
||||
```json
|
||||
{
|
||||
"model": "meta-llama/Llama-3.1-8B-Instruct",
|
||||
"input": "Tell me a joke."
|
||||
}
|
||||
```
|
||||
|
||||
#### Anthropic Messages API
|
||||
|
||||
**Path:** `/anthropic/v1/messages`
|
||||
|
||||
Supports the [Anthropic Messages API](https://docs.anthropic.com/en/api/messages) format. Served directly, bypassing the RunPod queue.
|
||||
|
||||
```json
|
||||
{
|
||||
"model": "meta-llama/Llama-3.1-8B-Instruct",
|
||||
"max_tokens": 256,
|
||||
"messages": [
|
||||
{"role": "user", "content": "Hello!"}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
#### Response Format
|
||||
|
||||
Both APIs return the same response format:
|
||||
@@ -180,7 +209,7 @@ Both APIs return the same response format:
|
||||
|
||||
Below are minimal `python` snippets so you can copy-paste to get started quickly.
|
||||
|
||||
> Replace `<ENDPOINT_ID>` with your endpoint ID and `<API_KEY>` with a [Runpod API key](https://docs.runpod.io/get-started/api-keys).
|
||||
> Replace `<ENDPOINT_ID>` with your endpoint ID and `<API_KEY>` with a [RunPod API key](https://docs.runpod.io/get-started/api-keys).
|
||||
|
||||
### OpenAI compatible API
|
||||
|
||||
|
||||
+1
-6
@@ -2,7 +2,7 @@ FROM nvidia/cuda:12.9.1-base-ubuntu22.04
|
||||
|
||||
RUN apt-get update -y \
|
||||
&& apt-get install -y python3-pip curl \
|
||||
&& curl -LsSf https://astral.sh/uv/install.sh | sh
|
||||
&& curl -LsSf https://astral.sh/uv/0.10.9/install.sh | sh
|
||||
|
||||
ENV PATH="/root/.local/bin:$PATH"
|
||||
|
||||
@@ -25,7 +25,6 @@ ARG QUANTIZATION=""
|
||||
ARG MODEL_REVISION=""
|
||||
ARG TOKENIZER_REVISION=""
|
||||
ARG VLLM_NIGHTLY="false"
|
||||
ARG LMCACHE="false"
|
||||
|
||||
ENV MODEL_NAME=$MODEL_NAME \
|
||||
MODEL_REVISION=$MODEL_REVISION \
|
||||
@@ -47,10 +46,6 @@ ENV MODEL_NAME=$MODEL_NAME \
|
||||
|
||||
ENV PYTHONPATH="/:/vllm-workspace"
|
||||
|
||||
RUN if [ "${LMCACHE}" = "true" ]; then \
|
||||
uv pip install --system lmcache; \
|
||||
fi
|
||||
|
||||
RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \
|
||||
uv pip install --system -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \
|
||||
apt-get update && apt-get install -y git && rm -rf /var/lib/apt/lists/* && \
|
||||
|
||||
@@ -3,12 +3,13 @@ pandas
|
||||
pyarrow
|
||||
runpod
|
||||
huggingface-hub
|
||||
lmcache==0.4.2
|
||||
packaging>=24.2
|
||||
typing-extensions>=4.8.0
|
||||
pydantic
|
||||
pydantic-settings
|
||||
hf-transfer
|
||||
transformers>=5.2.0
|
||||
transformers>=5.2.0,<5.3.0
|
||||
bitsandbytes>=0.45.0
|
||||
kernels
|
||||
torch-c-dlpack-ext
|
||||
|
||||
+101
-12
@@ -384,36 +384,112 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
yield batch
|
||||
|
||||
async def _handle_responses_request(self, openai_request: JobInput):
|
||||
request_id = getattr(openai_request, "request_id", "unknown")
|
||||
|
||||
try:
|
||||
request = ResponsesRequest(**openai_request.openai_input)
|
||||
except Exception as e:
|
||||
yield create_error_response(str(e)).model_dump()
|
||||
logging.error(
|
||||
"Invalid ResponsesRequest JSON: %s",
|
||||
e,
|
||||
extra={"request_id": request_id}
|
||||
)
|
||||
yield create_error_response(
|
||||
"Invalid request format",
|
||||
err_type="BadRequestError"
|
||||
).model_dump()
|
||||
return
|
||||
|
||||
dummy_request = DummyRequest()
|
||||
response = await self.responses_engine.create_responses(request, raw_request=dummy_request)
|
||||
try:
|
||||
response = await self.responses_engine.create_responses(request, raw_request=dummy_request)
|
||||
except Exception as e:
|
||||
logging.error(
|
||||
"Failed to create Responses: %s",
|
||||
e,
|
||||
extra={"request_id": request_id},
|
||||
exc_info=True
|
||||
)
|
||||
if isinstance(response, ErrorResponse):
|
||||
yield response.model_dump()
|
||||
else:
|
||||
yield create_error_response(
|
||||
"Internal server error during response generation",
|
||||
err_type="InternalServerError"
|
||||
).model_dump()
|
||||
return
|
||||
|
||||
if isinstance(response, (ErrorResponse, ResponsesResponse)):
|
||||
yield response.model_dump()
|
||||
return
|
||||
|
||||
async for event in response:
|
||||
event_type = getattr(event, "type", "unknown")
|
||||
yield f"event: {event_type}\ndata: {event.model_dump_json(indent=None)}\n\n"
|
||||
|
||||
try:
|
||||
async for event in response:
|
||||
if not hasattr(event, "type"):
|
||||
continue
|
||||
event_type = getattr(event, "type", "unknown")
|
||||
yield f"event: {event_type}\ndata: {event.model_dump_json(indent=None)}\n\n"
|
||||
except Exception as e:
|
||||
logging.error(
|
||||
"Error processing responses stream: %s",
|
||||
e,
|
||||
extra={"request_id": request_id},
|
||||
exc_info=True
|
||||
)
|
||||
yield create_error_response(
|
||||
"Streaming response failed",
|
||||
err_type="InternalServerError"
|
||||
).model_dump()
|
||||
async def _handle_messages_request(self, openai_request: JobInput):
|
||||
request_id = getattr(openai_request, "request_id", "unknown")
|
||||
|
||||
try:
|
||||
request = AnthropicMessagesRequest(**openai_request.openai_input)
|
||||
except Exception as e:
|
||||
yield create_error_response(str(e)).model_dump()
|
||||
logging.error(
|
||||
"Invalid AnthropicMessagesRequest: %s",
|
||||
e,
|
||||
extra={"request_id": request_id}
|
||||
)
|
||||
yield AnthropicErrorResponse(
|
||||
error=AnthropicError(
|
||||
type="invalid_request_error",
|
||||
message="Invalid request format"
|
||||
)
|
||||
).model_dump()
|
||||
return
|
||||
|
||||
dummy_request = DummyRequest()
|
||||
response = await self.messages_engine.create_messages(request, raw_request=dummy_request)
|
||||
|
||||
try:
|
||||
response = await self.messages_engine.create_messages(request, raw_request=dummy_request)
|
||||
except Exception as e:
|
||||
logging.error(
|
||||
"Failed to create messages: %s",
|
||||
e,
|
||||
extra={"request_id": request_id},
|
||||
exc_info=True
|
||||
)
|
||||
if isinstance(response, ErrorResponse):
|
||||
error_type = getattr(response, "type", "internal_error")
|
||||
error_message = getattr(response, "message", str(e)[:200])
|
||||
yield AnthropicErrorResponse(
|
||||
error=AnthropicError(type=error_type, message=error_message)
|
||||
).model_dump()
|
||||
else:
|
||||
yield AnthropicErrorResponse(
|
||||
error=AnthropicError(
|
||||
type="internal_error",
|
||||
message="Failed to generate messages"
|
||||
)
|
||||
).model_dump()
|
||||
return
|
||||
|
||||
if isinstance(response, ErrorResponse):
|
||||
error_type = getattr(response, "type", "internal_error")
|
||||
error_message = getattr(response, "message", "Unknown error")
|
||||
yield AnthropicErrorResponse(
|
||||
error=AnthropicError(type=response.error.type, message=response.error.message)
|
||||
error=AnthropicError(type=error_type, message=error_message)
|
||||
).model_dump()
|
||||
return
|
||||
|
||||
@@ -421,6 +497,19 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
yield response.model_dump(exclude_none=True)
|
||||
return
|
||||
|
||||
async for chunk in response:
|
||||
yield chunk
|
||||
|
||||
try:
|
||||
async for chunk in response:
|
||||
yield chunk
|
||||
except Exception as e:
|
||||
logging.error(
|
||||
"Error streaming messages: %s",
|
||||
e,
|
||||
extra={"request_id": request_id},
|
||||
exc_info=True
|
||||
)
|
||||
yield AnthropicErrorResponse(
|
||||
error=AnthropicError(
|
||||
type="internal_error",
|
||||
message="Error while streaming messages"
|
||||
)
|
||||
).model_dump()
|
||||
|
||||
+26
-9
@@ -402,15 +402,32 @@ def get_engine_args():
|
||||
logging.warning("Overriding MAX_PARALLEL_LOADING_WORKERS with None because more than 1 GPU is available.")
|
||||
|
||||
# LMCache requires HMA to be disabled
|
||||
_kv_transfer = args.get("kv_transfer_config")
|
||||
_kv_offload = args.get("kv_offloading_backend")
|
||||
_lmcache_active = _kv_offload == "lmcache" or (
|
||||
isinstance(_kv_transfer, dict)
|
||||
and "lmcache" in str(_kv_transfer.get("kv_connector", "")).lower()
|
||||
)
|
||||
if _lmcache_active and not args.get("disable_hybrid_kv_cache_manager"):
|
||||
args["disable_hybrid_kv_cache_manager"] = True
|
||||
logging.info("LMCache detected: automatically setting disable_hybrid_kv_cache_manager=True")
|
||||
try:
|
||||
_kv_transfer = args.get("kv_transfer_config")
|
||||
_kv_offload = args.get("kv_offloading_backend")
|
||||
|
||||
lmcache_detected = _kv_offload == "lmcache" or (
|
||||
isinstance(_kv_transfer, dict)
|
||||
and isinstance(_kv_transfer.get("kv_connector"), str)
|
||||
and "lmcache" in _kv_transfer.get("kv_connector", "").lower()
|
||||
)
|
||||
|
||||
if lmcache_detected and not args.get("disable_hybrid_kv_cache_manager"):
|
||||
args["disable_hybrid_kv_cache_manager"] = True
|
||||
args["kv_offloading_backend"] = None
|
||||
args["kv_transfer_config"] = None
|
||||
logging.info("LMCache detected: automatically setting disable_hybrid_kv_cache_manager=True and clearing conflicting settings")
|
||||
elif lmcache_detected and args.get("disable_hybrid_kv_cache_manager") is False:
|
||||
logging.warning(
|
||||
"LMCache configuration detected but disabled: "
|
||||
"disable_hybrid_kv_cache_manager must be False when using LMCache"
|
||||
)
|
||||
except Exception as e:
|
||||
logging.error(
|
||||
"Failed to check LMCache configuration: %s",
|
||||
e,
|
||||
exc_info=True
|
||||
)
|
||||
|
||||
# Deprecated env args backwards compatibility
|
||||
if args.get("kv_cache_dtype") == "fp8_e5m2":
|
||||
|
||||
Reference in New Issue
Block a user