Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2df915a145 | ||
|
|
aadc025849 | ||
|
|
8b4a49073d | ||
|
|
4e10641d69 | ||
|
|
6e8696c12a | ||
|
|
b49e81a75a | ||
|
|
65932f85e1 | ||
|
|
94840cfbbb | ||
|
|
677a01e8f3 | ||
|
|
5cd12ba331 |
+1
-1
@@ -12,7 +12,7 @@ RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
python3 -m pip install --upgrade -r /requirements.txt
|
||||
|
||||
# Install vLLM (switching back to pip installs since issues that required building fork are fixed and space optimization is not as important since caching) and FlashInfer
|
||||
RUN python3 -m pip install vllm==0.6.3 && \
|
||||
RUN python3 -m pip install vllm==0.6.4 && \
|
||||
python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3
|
||||
|
||||
# Setup for Option 2: Building the Image with the Model included
|
||||
|
||||
@@ -18,9 +18,9 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
|
||||
### 1. UI for Deploying vLLM Worker on RunPod console:
|
||||

|
||||
|
||||
### 2. Worker vLLM `v1.5.0` with vLLM `0.6.2` now available under `stable` tags
|
||||
### 2. Worker vLLM `v1.6.0` with vLLM `0.6.3` now available under `stable` tags
|
||||
|
||||
Update v1.5.0 is now available, use the image tag `runpod/worker-v1-vllm:v1.5.0stable-cuda12.1.0`.
|
||||
Update v1.6.0 is now available, use the image tag `runpod/worker-v1-vllm:v1.6.0stable-cuda12.1.0`.
|
||||
|
||||
### 3. OpenAI-Compatible [Embedding Worker](https://github.com/runpod-workers/worker-infinity-embedding) Released
|
||||
Deploy your own OpenAI-compatible Serverless Endpoint on RunPod with multiple embedding models and fast inference for RAG and more!
|
||||
@@ -82,7 +82,7 @@ Below is a summary of the available RunPod Worker images, categorized by image s
|
||||
|
||||
| CUDA Version | Stable Image Tag | Development Image Tag | Note |
|
||||
|--------------|-----------------------------------|-----------------------------------|----------------------------------------------------------------------|
|
||||
| 12.1.0 | `runpod/worker-v1-vllm:v1.5.0stable-cuda12.1.0` | `runpod/worker-v1-vllm:v1.5.0dev-cuda12.1.0` | When creating an Endpoint, select CUDA Version 12.3, 12.2 and 12.1 in the filter. |
|
||||
| 12.1.0 | `runpod/worker-v1-vllm:v1.6.0stable-cuda12.1.0` | `runpod/worker-v1-vllm:v1.6.0dev-cuda12.1.0` | When creating an Endpoint, select CUDA Version 12.3, 12.2 and 12.1 in the filter. |
|
||||
|
||||
|
||||
|
||||
|
||||
+15
-6
@@ -11,7 +11,8 @@ from vllm import AsyncLLMEngine
|
||||
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
|
||||
from vllm.entrypoints.openai.serving_completion import OpenAIServingCompletion
|
||||
from vllm.entrypoints.openai.protocol import ChatCompletionRequest, CompletionRequest, ErrorResponse
|
||||
from vllm.entrypoints.openai.serving_engine import BaseModelPath
|
||||
from vllm.entrypoints.openai.serving_engine import BaseModelPath, LoRAModulePath
|
||||
|
||||
|
||||
from utils import DummyRequest, JobInput, BatchSize, create_error_response
|
||||
from constants import DEFAULT_MAX_CONCURRENCY, DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MIN_BATCH_SIZE
|
||||
@@ -128,13 +129,24 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
self.base_model_paths = [
|
||||
BaseModelPath(name=self.engine_args.model, model_path=self.engine_args.model)
|
||||
]
|
||||
|
||||
lora_modules = os.getenv('LORA_MODULES', None)
|
||||
if lora_modules is not None:
|
||||
try:
|
||||
lora_modules = json.loads(lora_modules)
|
||||
lora_modules = [LoRAModulePath(**lora_modules)]
|
||||
except:
|
||||
lora_modules = None
|
||||
|
||||
|
||||
|
||||
self.chat_engine = OpenAIServingChat(
|
||||
engine_client=self.llm,
|
||||
model_config=self.model_config,
|
||||
base_model_paths=self.base_model_paths,
|
||||
response_role=self.response_role,
|
||||
chat_template=self.tokenizer.tokenizer.chat_template,
|
||||
lora_modules=None,
|
||||
lora_modules=lora_modules,
|
||||
prompt_adapters=None,
|
||||
request_logger=None
|
||||
)
|
||||
@@ -142,7 +154,7 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
engine_client=self.llm,
|
||||
model_config=self.model_config,
|
||||
base_model_paths=self.base_model_paths,
|
||||
lora_modules=[],
|
||||
lora_modules=lora_modules,
|
||||
prompt_adapters=None,
|
||||
request_logger=None
|
||||
)
|
||||
@@ -158,9 +170,6 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
|
||||
async def _handle_model_request(self):
|
||||
models = await self.chat_engine.show_available_models()
|
||||
fixed_model = models.data[0]
|
||||
fixed_model.id = self.served_model_name
|
||||
models.data = [fixed_model]
|
||||
return models.model_dump()
|
||||
|
||||
async def _handle_chat_or_completion_request(self, openai_request: JobInput):
|
||||
|
||||
+955
-895
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user