requested changes

This commit is contained in:
velaraptor-runpod
2026-04-03 15:53:40 -05:00
parent 30e8514d63
commit 3ef1fb8e7b
5 changed files with 165 additions and 34 deletions
+35 -6
View File
@@ -4,7 +4,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
---
[![Runpod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
[![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
---
@@ -34,11 +34,11 @@ For complete configuration options, see the [full configuration documentation](h
## API Usage
This worker supports two API formats: **Runpod native** and **OpenAI-compatible**.
This worker supports two API formats: **RunPod native** and **OpenAI-compatible**.
### Runpod Native API
### RunPod Native API
For testing directly in the Runpod UI, use these examples in your endpoint's request tab.
For testing directly in the RunPod UI, use these examples in your endpoint's request tab.
#### Chat Completions
@@ -104,7 +104,7 @@ For direct text generation without chat format:
### OpenAI-Compatible API
For external clients and SDKs, use the `/openai/v1` path prefix with your Runpod API key.
For external clients and SDKs, use the `/openai/v1` path prefix with your RunPod API key.
#### Chat Completions
@@ -157,6 +157,35 @@ For external clients and SDKs, use the `/openai/v1` path prefix with your Runpod
{}
```
#### OpenAI Responses API
**Path:** `/openai/v1/responses`
Supports the [OpenAI Responses API](https://platform.openai.com/docs/api-reference/responses) format. Note: this route bypasses the RunPod queue and is served directly — use `/openai/` prefixed paths rather than the RunPod job queue for these endpoints.
```json
{
"model": "meta-llama/Llama-3.1-8B-Instruct",
"input": "Tell me a joke."
}
```
#### Anthropic Messages API
**Path:** `/anthropic/v1/messages`
Supports the [Anthropic Messages API](https://docs.anthropic.com/en/api/messages) format. Served directly, bypassing the RunPod queue.
```json
{
"model": "meta-llama/Llama-3.1-8B-Instruct",
"max_tokens": 256,
"messages": [
{"role": "user", "content": "Hello!"}
]
}
```
#### Response Format
Both APIs return the same response format:
@@ -180,7 +209,7 @@ Both APIs return the same response format:
Below are minimal `python` snippets so you can copy-paste to get started quickly.
> Replace `<ENDPOINT_ID>` with your endpoint ID and `<API_KEY>` with a [Runpod API key](https://docs.runpod.io/get-started/api-keys).
> Replace `<ENDPOINT_ID>` with your endpoint ID and `<API_KEY>` with a [RunPod API key](https://docs.runpod.io/get-started/api-keys).
### OpenAI compatible API
+1 -6
View File
@@ -2,7 +2,7 @@ FROM nvidia/cuda:12.9.1-base-ubuntu22.04
RUN apt-get update -y \
&& apt-get install -y python3-pip curl \
&& curl -LsSf https://astral.sh/uv/install.sh | sh
&& curl -LsSf https://astral.sh/uv/0.10.9/install.sh | sh
ENV PATH="/root/.local/bin:$PATH"
@@ -25,7 +25,6 @@ ARG QUANTIZATION=""
ARG MODEL_REVISION=""
ARG TOKENIZER_REVISION=""
ARG VLLM_NIGHTLY="false"
ARG LMCACHE="false"
ENV MODEL_NAME=$MODEL_NAME \
MODEL_REVISION=$MODEL_REVISION \
@@ -47,10 +46,6 @@ ENV MODEL_NAME=$MODEL_NAME \
ENV PYTHONPATH="/:/vllm-workspace"
RUN if [ "${LMCACHE}" = "true" ]; then \
uv pip install --system lmcache; \
fi
RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \
uv pip install --system -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \
apt-get update && apt-get install -y git && rm -rf /var/lib/apt/lists/* && \
+2 -1
View File
@@ -3,12 +3,13 @@ pandas
pyarrow
runpod
huggingface-hub
lmcache==0.4.2
packaging>=24.2
typing-extensions>=4.8.0
pydantic
pydantic-settings
hf-transfer
transformers>=5.2.0
transformers>=5.2.0,<5.3.0
bitsandbytes>=0.45.0
kernels
torch-c-dlpack-ext
+101 -12
View File
@@ -384,36 +384,112 @@ class OpenAIvLLMEngine(vLLMEngine):
yield batch
async def _handle_responses_request(self, openai_request: JobInput):
request_id = getattr(openai_request, "request_id", "unknown")
try:
request = ResponsesRequest(**openai_request.openai_input)
except Exception as e:
yield create_error_response(str(e)).model_dump()
logging.error(
"Invalid ResponsesRequest JSON: %s",
e,
extra={"request_id": request_id}
)
yield create_error_response(
"Invalid request format",
err_type="BadRequestError"
).model_dump()
return
dummy_request = DummyRequest()
response = await self.responses_engine.create_responses(request, raw_request=dummy_request)
try:
response = await self.responses_engine.create_responses(request, raw_request=dummy_request)
except Exception as e:
logging.error(
"Failed to create Responses: %s",
e,
extra={"request_id": request_id},
exc_info=True
)
if isinstance(response, ErrorResponse):
yield response.model_dump()
else:
yield create_error_response(
"Internal server error during response generation",
err_type="InternalServerError"
).model_dump()
return
if isinstance(response, (ErrorResponse, ResponsesResponse)):
yield response.model_dump()
return
async for event in response:
event_type = getattr(event, "type", "unknown")
yield f"event: {event_type}\ndata: {event.model_dump_json(indent=None)}\n\n"
try:
async for event in response:
if not hasattr(event, "type"):
continue
event_type = getattr(event, "type", "unknown")
yield f"event: {event_type}\ndata: {event.model_dump_json(indent=None)}\n\n"
except Exception as e:
logging.error(
"Error processing responses stream: %s",
e,
extra={"request_id": request_id},
exc_info=True
)
yield create_error_response(
"Streaming response failed",
err_type="InternalServerError"
).model_dump()
async def _handle_messages_request(self, openai_request: JobInput):
request_id = getattr(openai_request, "request_id", "unknown")
try:
request = AnthropicMessagesRequest(**openai_request.openai_input)
except Exception as e:
yield create_error_response(str(e)).model_dump()
logging.error(
"Invalid AnthropicMessagesRequest: %s",
e,
extra={"request_id": request_id}
)
yield AnthropicErrorResponse(
error=AnthropicError(
type="invalid_request_error",
message="Invalid request format"
)
).model_dump()
return
dummy_request = DummyRequest()
response = await self.messages_engine.create_messages(request, raw_request=dummy_request)
try:
response = await self.messages_engine.create_messages(request, raw_request=dummy_request)
except Exception as e:
logging.error(
"Failed to create messages: %s",
e,
extra={"request_id": request_id},
exc_info=True
)
if isinstance(response, ErrorResponse):
error_type = getattr(response, "type", "internal_error")
error_message = getattr(response, "message", str(e)[:200])
yield AnthropicErrorResponse(
error=AnthropicError(type=error_type, message=error_message)
).model_dump()
else:
yield AnthropicErrorResponse(
error=AnthropicError(
type="internal_error",
message="Failed to generate messages"
)
).model_dump()
return
if isinstance(response, ErrorResponse):
error_type = getattr(response, "type", "internal_error")
error_message = getattr(response, "message", "Unknown error")
yield AnthropicErrorResponse(
error=AnthropicError(type=response.error.type, message=response.error.message)
error=AnthropicError(type=error_type, message=error_message)
).model_dump()
return
@@ -421,6 +497,19 @@ class OpenAIvLLMEngine(vLLMEngine):
yield response.model_dump(exclude_none=True)
return
async for chunk in response:
yield chunk
try:
async for chunk in response:
yield chunk
except Exception as e:
logging.error(
"Error streaming messages: %s",
e,
extra={"request_id": request_id},
exc_info=True
)
yield AnthropicErrorResponse(
error=AnthropicError(
type="internal_error",
message="Error while streaming messages"
)
).model_dump()
+26 -9
View File
@@ -402,15 +402,32 @@ def get_engine_args():
logging.warning("Overriding MAX_PARALLEL_LOADING_WORKERS with None because more than 1 GPU is available.")
# LMCache requires HMA to be disabled
_kv_transfer = args.get("kv_transfer_config")
_kv_offload = args.get("kv_offloading_backend")
_lmcache_active = _kv_offload == "lmcache" or (
isinstance(_kv_transfer, dict)
and "lmcache" in str(_kv_transfer.get("kv_connector", "")).lower()
)
if _lmcache_active and not args.get("disable_hybrid_kv_cache_manager"):
args["disable_hybrid_kv_cache_manager"] = True
logging.info("LMCache detected: automatically setting disable_hybrid_kv_cache_manager=True")
try:
_kv_transfer = args.get("kv_transfer_config")
_kv_offload = args.get("kv_offloading_backend")
lmcache_detected = _kv_offload == "lmcache" or (
isinstance(_kv_transfer, dict)
and isinstance(_kv_transfer.get("kv_connector"), str)
and "lmcache" in _kv_transfer.get("kv_connector", "").lower()
)
if lmcache_detected and not args.get("disable_hybrid_kv_cache_manager"):
args["disable_hybrid_kv_cache_manager"] = True
args["kv_offloading_backend"] = None
args["kv_transfer_config"] = None
logging.info("LMCache detected: automatically setting disable_hybrid_kv_cache_manager=True and clearing conflicting settings")
elif lmcache_detected and args.get("disable_hybrid_kv_cache_manager") is False:
logging.warning(
"LMCache configuration detected but disabled: "
"disable_hybrid_kv_cache_manager must be False when using LMCache"
)
except Exception as e:
logging.error(
"Failed to check LMCache configuration: %s",
e,
exc_info=True
)
# Deprecated env args backwards compatibility
if args.get("kv_cache_dtype") == "fp8_e5m2":