From 3ef1fb8e7b4357dbe46638181b2db76759f20219 Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Fri, 3 Apr 2026 15:53:40 -0500 Subject: [PATCH] requested changes --- .runpod/README.md | 41 +++++++++++--- Dockerfile | 7 +-- builder/requirements.txt | 3 +- src/engine.py | 113 ++++++++++++++++++++++++++++++++++----- src/engine_args.py | 35 ++++++++---- 5 files changed, 165 insertions(+), 34 deletions(-) diff --git a/.runpod/README.md b/.runpod/README.md index 5f6e4d9..4fed299 100644 --- a/.runpod/README.md +++ b/.runpod/README.md @@ -4,7 +4,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API --- -[![Runpod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) +[![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm) --- @@ -34,11 +34,11 @@ For complete configuration options, see the [full configuration documentation](h ## API Usage -This worker supports two API formats: **Runpod native** and **OpenAI-compatible**. +This worker supports two API formats: **RunPod native** and **OpenAI-compatible**. -### Runpod Native API +### RunPod Native API -For testing directly in the Runpod UI, use these examples in your endpoint's request tab. +For testing directly in the RunPod UI, use these examples in your endpoint's request tab. #### Chat Completions @@ -104,7 +104,7 @@ For direct text generation without chat format: ### OpenAI-Compatible API -For external clients and SDKs, use the `/openai/v1` path prefix with your Runpod API key. +For external clients and SDKs, use the `/openai/v1` path prefix with your RunPod API key. #### Chat Completions @@ -157,6 +157,35 @@ For external clients and SDKs, use the `/openai/v1` path prefix with your Runpod {} ``` +#### OpenAI Responses API + +**Path:** `/openai/v1/responses` + +Supports the [OpenAI Responses API](https://platform.openai.com/docs/api-reference/responses) format. Note: this route bypasses the RunPod queue and is served directly — use `/openai/` prefixed paths rather than the RunPod job queue for these endpoints. + +```json +{ + "model": "meta-llama/Llama-3.1-8B-Instruct", + "input": "Tell me a joke." +} +``` + +#### Anthropic Messages API + +**Path:** `/anthropic/v1/messages` + +Supports the [Anthropic Messages API](https://docs.anthropic.com/en/api/messages) format. Served directly, bypassing the RunPod queue. + +```json +{ + "model": "meta-llama/Llama-3.1-8B-Instruct", + "max_tokens": 256, + "messages": [ + {"role": "user", "content": "Hello!"} + ] +} +``` + #### Response Format Both APIs return the same response format: @@ -180,7 +209,7 @@ Both APIs return the same response format: Below are minimal `python` snippets so you can copy-paste to get started quickly. -> Replace `` with your endpoint ID and `` with a [Runpod API key](https://docs.runpod.io/get-started/api-keys). +> Replace `` with your endpoint ID and `` with a [RunPod API key](https://docs.runpod.io/get-started/api-keys). ### OpenAI compatible API diff --git a/Dockerfile b/Dockerfile index bcf4069..6e103ab 100644 --- a/Dockerfile +++ b/Dockerfile @@ -2,7 +2,7 @@ FROM nvidia/cuda:12.9.1-base-ubuntu22.04 RUN apt-get update -y \ && apt-get install -y python3-pip curl \ - && curl -LsSf https://astral.sh/uv/install.sh | sh + && curl -LsSf https://astral.sh/uv/0.10.9/install.sh | sh ENV PATH="/root/.local/bin:$PATH" @@ -25,7 +25,6 @@ ARG QUANTIZATION="" ARG MODEL_REVISION="" ARG TOKENIZER_REVISION="" ARG VLLM_NIGHTLY="false" -ARG LMCACHE="false" ENV MODEL_NAME=$MODEL_NAME \ MODEL_REVISION=$MODEL_REVISION \ @@ -47,10 +46,6 @@ ENV MODEL_NAME=$MODEL_NAME \ ENV PYTHONPATH="/:/vllm-workspace" -RUN if [ "${LMCACHE}" = "true" ]; then \ - uv pip install --system lmcache; \ -fi - RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \ uv pip install --system -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \ apt-get update && apt-get install -y git && rm -rf /var/lib/apt/lists/* && \ diff --git a/builder/requirements.txt b/builder/requirements.txt index cd7c13d..cd3d924 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -3,12 +3,13 @@ pandas pyarrow runpod huggingface-hub +lmcache==0.4.2 packaging>=24.2 typing-extensions>=4.8.0 pydantic pydantic-settings hf-transfer -transformers>=5.2.0 +transformers>=5.2.0,<5.3.0 bitsandbytes>=0.45.0 kernels torch-c-dlpack-ext diff --git a/src/engine.py b/src/engine.py index 40a0296..c60dc3b 100644 --- a/src/engine.py +++ b/src/engine.py @@ -384,36 +384,112 @@ class OpenAIvLLMEngine(vLLMEngine): yield batch async def _handle_responses_request(self, openai_request: JobInput): + request_id = getattr(openai_request, "request_id", "unknown") + try: request = ResponsesRequest(**openai_request.openai_input) except Exception as e: - yield create_error_response(str(e)).model_dump() + logging.error( + "Invalid ResponsesRequest JSON: %s", + e, + extra={"request_id": request_id} + ) + yield create_error_response( + "Invalid request format", + err_type="BadRequestError" + ).model_dump() return dummy_request = DummyRequest() - response = await self.responses_engine.create_responses(request, raw_request=dummy_request) + try: + response = await self.responses_engine.create_responses(request, raw_request=dummy_request) + except Exception as e: + logging.error( + "Failed to create Responses: %s", + e, + extra={"request_id": request_id}, + exc_info=True + ) + if isinstance(response, ErrorResponse): + yield response.model_dump() + else: + yield create_error_response( + "Internal server error during response generation", + err_type="InternalServerError" + ).model_dump() + return if isinstance(response, (ErrorResponse, ResponsesResponse)): yield response.model_dump() return - async for event in response: - event_type = getattr(event, "type", "unknown") - yield f"event: {event_type}\ndata: {event.model_dump_json(indent=None)}\n\n" - + try: + async for event in response: + if not hasattr(event, "type"): + continue + event_type = getattr(event, "type", "unknown") + yield f"event: {event_type}\ndata: {event.model_dump_json(indent=None)}\n\n" + except Exception as e: + logging.error( + "Error processing responses stream: %s", + e, + extra={"request_id": request_id}, + exc_info=True + ) + yield create_error_response( + "Streaming response failed", + err_type="InternalServerError" + ).model_dump() async def _handle_messages_request(self, openai_request: JobInput): + request_id = getattr(openai_request, "request_id", "unknown") + try: request = AnthropicMessagesRequest(**openai_request.openai_input) except Exception as e: - yield create_error_response(str(e)).model_dump() + logging.error( + "Invalid AnthropicMessagesRequest: %s", + e, + extra={"request_id": request_id} + ) + yield AnthropicErrorResponse( + error=AnthropicError( + type="invalid_request_error", + message="Invalid request format" + ) + ).model_dump() return dummy_request = DummyRequest() - response = await self.messages_engine.create_messages(request, raw_request=dummy_request) + + try: + response = await self.messages_engine.create_messages(request, raw_request=dummy_request) + except Exception as e: + logging.error( + "Failed to create messages: %s", + e, + extra={"request_id": request_id}, + exc_info=True + ) + if isinstance(response, ErrorResponse): + error_type = getattr(response, "type", "internal_error") + error_message = getattr(response, "message", str(e)[:200]) + yield AnthropicErrorResponse( + error=AnthropicError(type=error_type, message=error_message) + ).model_dump() + else: + yield AnthropicErrorResponse( + error=AnthropicError( + type="internal_error", + message="Failed to generate messages" + ) + ).model_dump() + return if isinstance(response, ErrorResponse): + error_type = getattr(response, "type", "internal_error") + error_message = getattr(response, "message", "Unknown error") yield AnthropicErrorResponse( - error=AnthropicError(type=response.error.type, message=response.error.message) + error=AnthropicError(type=error_type, message=error_message) ).model_dump() return @@ -421,6 +497,19 @@ class OpenAIvLLMEngine(vLLMEngine): yield response.model_dump(exclude_none=True) return - async for chunk in response: - yield chunk - \ No newline at end of file + try: + async for chunk in response: + yield chunk + except Exception as e: + logging.error( + "Error streaming messages: %s", + e, + extra={"request_id": request_id}, + exc_info=True + ) + yield AnthropicErrorResponse( + error=AnthropicError( + type="internal_error", + message="Error while streaming messages" + ) + ).model_dump() diff --git a/src/engine_args.py b/src/engine_args.py index 05dd2b7..1fdbe71 100644 --- a/src/engine_args.py +++ b/src/engine_args.py @@ -402,15 +402,32 @@ def get_engine_args(): logging.warning("Overriding MAX_PARALLEL_LOADING_WORKERS with None because more than 1 GPU is available.") # LMCache requires HMA to be disabled - _kv_transfer = args.get("kv_transfer_config") - _kv_offload = args.get("kv_offloading_backend") - _lmcache_active = _kv_offload == "lmcache" or ( - isinstance(_kv_transfer, dict) - and "lmcache" in str(_kv_transfer.get("kv_connector", "")).lower() - ) - if _lmcache_active and not args.get("disable_hybrid_kv_cache_manager"): - args["disable_hybrid_kv_cache_manager"] = True - logging.info("LMCache detected: automatically setting disable_hybrid_kv_cache_manager=True") + try: + _kv_transfer = args.get("kv_transfer_config") + _kv_offload = args.get("kv_offloading_backend") + + lmcache_detected = _kv_offload == "lmcache" or ( + isinstance(_kv_transfer, dict) + and isinstance(_kv_transfer.get("kv_connector"), str) + and "lmcache" in _kv_transfer.get("kv_connector", "").lower() + ) + + if lmcache_detected and not args.get("disable_hybrid_kv_cache_manager"): + args["disable_hybrid_kv_cache_manager"] = True + args["kv_offloading_backend"] = None + args["kv_transfer_config"] = None + logging.info("LMCache detected: automatically setting disable_hybrid_kv_cache_manager=True and clearing conflicting settings") + elif lmcache_detected and args.get("disable_hybrid_kv_cache_manager") is False: + logging.warning( + "LMCache configuration detected but disabled: " + "disable_hybrid_kv_cache_manager must be False when using LMCache" + ) + except Exception as e: + logging.error( + "Failed to check LMCache configuration: %s", + e, + exc_info=True + ) # Deprecated env args backwards compatibility if args.get("kv_cache_dtype") == "fp8_e5m2":