Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4395d0c67b | ||
|
|
5864fa6843 | ||
|
|
067f0bc173 | ||
|
|
0869068e7c | ||
|
|
19f25b17de | ||
|
|
70aa748c30 | ||
|
|
da3e5f524b | ||
|
|
acbdf63de8 | ||
|
|
fa31d9663d | ||
|
|
b8fb313483 | ||
|
|
9635daf336 | ||
|
|
fb4e700090 | ||
|
|
d4872ed9c8 | ||
|
|
b2eef50c3b | ||
|
|
8005bcc1a8 | ||
|
|
aa7b00ddda | ||
|
|
dc6f3239bd | ||
|
|
f9d0fcb78c | ||
|
|
99b952e55e | ||
|
|
389fad7952 | ||
|
|
56dc4ad075 | ||
|
|
6dcf39e159 | ||
|
|
2b1d618287 | ||
|
|
d7e9c49fe4 | ||
|
|
c9791f1163 | ||
|
|
6fc770415d | ||
|
|
9e8d9196b0 | ||
|
|
30dd7c1eb5 | ||
|
|
aa7c37ffb0 | ||
|
|
dc8c88027a | ||
|
|
a948e90caa | ||
|
|
c703254f71 | ||
|
|
2747106403 | ||
|
|
91ed30e9a2 | ||
|
|
fcbfe84f63 | ||
|
|
610429df23 | ||
|
|
a578c6df23 | ||
|
|
0e3359a70e | ||
|
|
3f0a20d28e | ||
|
|
8e3c26be14 | ||
|
|
66ea8b1110 | ||
|
|
a3d432afdf | ||
|
|
06c2bb1715 | ||
|
|
d2e355eae9 | ||
|
|
b7787d8ca8 | ||
|
|
b1fca5d257 | ||
|
|
3e86d16892 | ||
|
|
d9f54ce76a | ||
|
|
149da95cd0 | ||
|
|
0a89394f1d | ||
|
|
8df7f41f1d | ||
|
|
4d7b8c03c0 | ||
|
|
6c6bf50379 | ||
|
|
27a2ee5754 | ||
|
|
2df915a145 | ||
|
|
aadc025849 | ||
|
|
8b4a49073d | ||
|
|
4e10641d69 | ||
|
|
6e8696c12a | ||
|
|
b49e81a75a | ||
|
|
65932f85e1 | ||
|
|
94840cfbbb | ||
|
|
ce47c41f4a | ||
|
|
c03ecc42fe | ||
|
|
ae56b9f43d | ||
|
|
891699be1e | ||
|
|
c28aa02576 | ||
|
|
eba20c0704 | ||
|
|
677a01e8f3 | ||
|
|
5cd12ba331 | ||
|
|
850c686538 | ||
|
|
de2876e659 |
@@ -19,32 +19,49 @@ jobs:
|
|||||||
|
|
||||||
- name: Check for new package version and update
|
- name: Check for new package version and update
|
||||||
run: |
|
run: |
|
||||||
# Get current version
|
echo "Fetching the current runpod version from requirements.txt..."
|
||||||
current_version=$(grep -oP 'runpod==\K[^"]+' ./builder/requirements.txt)
|
|
||||||
|
|
||||||
# Get new version
|
# Get current version, allowing both == and ~= in the search pattern
|
||||||
|
current_version=$(grep -oP 'runpod[~=]{1,2}\K[^"]+' ./builder/requirements.txt)
|
||||||
|
echo "Current version: $current_version"
|
||||||
|
|
||||||
|
# Extract major and minor from current version
|
||||||
|
current_major_minor=$(echo $current_version | cut -d. -f1,2)
|
||||||
|
echo "Current major.minor: $current_major_minor"
|
||||||
|
|
||||||
|
echo "Fetching the latest runpod version from PyPI..."
|
||||||
|
|
||||||
|
# Get new version from PyPI
|
||||||
new_version=$(curl -s https://pypi.org/pypi/runpod/json | jq -r .info.version)
|
new_version=$(curl -s https://pypi.org/pypi/runpod/json | jq -r .info.version)
|
||||||
echo "NEW_VERSION_ENV=$new_version" >> $GITHUB_ENV
|
echo "NEW_VERSION_ENV=$new_version" >> $GITHUB_ENV
|
||||||
|
echo "New version: $new_version"
|
||||||
|
|
||||||
|
# Extract major and minor from new version
|
||||||
|
new_major_minor=$(echo $new_version | cut -d. -f1,2)
|
||||||
|
echo "New major.minor: $new_major_minor"
|
||||||
|
|
||||||
if [ -z "$new_version" ]; then
|
if [ -z "$new_version" ]; then
|
||||||
echo "Failed to fetch the new version."
|
echo "ERROR: Failed to fetch the new version from PyPI."
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# Check if the version is already up-to-date
|
# Check if the major or minor version is different
|
||||||
if [ "$current_version" = "$new_version" ]; then
|
if [ "$current_major_minor" = "$new_major_minor" ]; then
|
||||||
echo "The package version is already up-to-date."
|
echo "No update needed. The new version ($new_major_minor) is within the allowed range (~= $current_major_minor)."
|
||||||
exit 0
|
exit 0
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# Update requirements.txt
|
echo "New major/minor detected ($new_major_minor). Updating requirements.txt..."
|
||||||
sed -i "s/runpod==.*/runpod==$new_version/" ./builder/requirements.txt
|
|
||||||
|
# Update requirements.txt, preserving the existing constraint type (~= or ==)
|
||||||
|
sed -i "s/runpod[~=][^ ]*/runpod~=$new_version/" ./builder/requirements.txt
|
||||||
|
echo "requirements.txt has been updated."
|
||||||
|
|
||||||
- name: Create Pull Request
|
- name: Create Pull Request
|
||||||
uses: peter-evans/create-pull-request@v3
|
uses: peter-evans/create-pull-request@v3
|
||||||
with:
|
with:
|
||||||
token: ${{ secrets.GITHUB_TOKEN }}
|
token: ${{ secrets.GITHUB_TOKEN }}
|
||||||
commit-message: Update package version
|
commit-message: Update runpod package version
|
||||||
title: Update runpod package version
|
title: Update runpod package version
|
||||||
body: The package version has been updated to ${{ env.NEW_VERSION_ENV }}
|
body: The package version has been updated to ${{ env.NEW_VERSION_ENV }}
|
||||||
branch: runpod-package-update
|
branch: runpod-package-update
|
||||||
|
|||||||
+1016
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,44 @@
|
|||||||
|
{
|
||||||
|
"tests": [
|
||||||
|
{
|
||||||
|
"name": "basic_inference_test",
|
||||||
|
"input": {
|
||||||
|
"prompt": "Write a short poem about artificial intelligence.",
|
||||||
|
},
|
||||||
|
"timeout": 30000
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"config": {
|
||||||
|
"gpuTypeId": "NVIDIA GeForce RTX 4090",
|
||||||
|
"gpuCount": 1,
|
||||||
|
"env": [
|
||||||
|
{
|
||||||
|
"key": "MODEL_NAME",
|
||||||
|
"value": "facebook/opt-350m"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"key": "HF_TOKEN",
|
||||||
|
"value": "hf_dummy_token_for_testing_purposes_only"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"key": "MAX_MODEL_LEN",
|
||||||
|
"value": "8192"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"key": "GPU_MEMORY_UTILIZATION",
|
||||||
|
"value": "0.95"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"allowedCudaVersions": [
|
||||||
|
"12.7",
|
||||||
|
"12.6",
|
||||||
|
"12.5",
|
||||||
|
"12.4",
|
||||||
|
"12.3",
|
||||||
|
"12.2",
|
||||||
|
"12.1",
|
||||||
|
"12.0",
|
||||||
|
"11.7"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
+4
-4
@@ -12,7 +12,7 @@ RUN --mount=type=cache,target=/root/.cache/pip \
|
|||||||
python3 -m pip install --upgrade -r /requirements.txt
|
python3 -m pip install --upgrade -r /requirements.txt
|
||||||
|
|
||||||
# Install vLLM (switching back to pip installs since issues that required building fork are fixed and space optimization is not as important since caching) and FlashInfer
|
# Install vLLM (switching back to pip installs since issues that required building fork are fixed and space optimization is not as important since caching) and FlashInfer
|
||||||
RUN python3 -m pip install vllm==0.6.2 && \
|
RUN python3 -m pip install vllm==0.8.2 && \
|
||||||
python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3
|
python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3
|
||||||
|
|
||||||
# Setup for Option 2: Building the Image with the Model included
|
# Setup for Option 2: Building the Image with the Model included
|
||||||
@@ -32,7 +32,7 @@ ENV MODEL_NAME=$MODEL_NAME \
|
|||||||
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
|
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
|
||||||
HUGGINGFACE_HUB_CACHE="${BASE_PATH}/huggingface-cache/hub" \
|
HUGGINGFACE_HUB_CACHE="${BASE_PATH}/huggingface-cache/hub" \
|
||||||
HF_HOME="${BASE_PATH}/huggingface-cache/hub" \
|
HF_HOME="${BASE_PATH}/huggingface-cache/hub" \
|
||||||
HF_HUB_ENABLE_HF_TRANSFER=1
|
HF_HUB_ENABLE_HF_TRANSFER=0
|
||||||
|
|
||||||
ENV PYTHONPATH="/:/vllm-workspace"
|
ENV PYTHONPATH="/:/vllm-workspace"
|
||||||
|
|
||||||
@@ -40,10 +40,10 @@ ENV PYTHONPATH="/:/vllm-workspace"
|
|||||||
COPY src /src
|
COPY src /src
|
||||||
RUN --mount=type=secret,id=HF_TOKEN,required=false \
|
RUN --mount=type=secret,id=HF_TOKEN,required=false \
|
||||||
if [ -f /run/secrets/HF_TOKEN ]; then \
|
if [ -f /run/secrets/HF_TOKEN ]; then \
|
||||||
export HF_TOKEN=$(cat /run/secrets/HF_TOKEN); \
|
export HF_TOKEN=$(cat /run/secrets/HF_TOKEN); \
|
||||||
fi && \
|
fi && \
|
||||||
if [ -n "$MODEL_NAME" ]; then \
|
if [ -n "$MODEL_NAME" ]; then \
|
||||||
python3 /src/download_model.py; \
|
python3 /src/download_model.py; \
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# Start the handler
|
# Start the handler
|
||||||
|
|||||||
@@ -18,9 +18,9 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
|
|||||||
### 1. UI for Deploying vLLM Worker on RunPod console:
|
### 1. UI for Deploying vLLM Worker on RunPod console:
|
||||||

|

|
||||||
|
|
||||||
### 2. Worker vLLM `v1.5.0` with vLLM `0.6.2` now available under `stable` tags
|
### 2. Worker vLLM `v2.2.0` with vLLM `0.8.2` now available under `stable` tags
|
||||||
|
|
||||||
Update v1.5.0 is now available, use the image tag `runpod/worker-v1-vllm:v1.5.0stable-cuda12.1.0`.
|
Update v2.2.0 is now available, use the image tag `runpod/worker-v1-vllm:v2.2.0stable-cuda12.1.0`.
|
||||||
|
|
||||||
### 3. OpenAI-Compatible [Embedding Worker](https://github.com/runpod-workers/worker-infinity-embedding) Released
|
### 3. OpenAI-Compatible [Embedding Worker](https://github.com/runpod-workers/worker-infinity-embedding) Released
|
||||||
Deploy your own OpenAI-compatible Serverless Endpoint on RunPod with multiple embedding models and fast inference for RAG and more!
|
Deploy your own OpenAI-compatible Serverless Endpoint on RunPod with multiple embedding models and fast inference for RAG and more!
|
||||||
@@ -82,7 +82,7 @@ Below is a summary of the available RunPod Worker images, categorized by image s
|
|||||||
|
|
||||||
| CUDA Version | Stable Image Tag | Development Image Tag | Note |
|
| CUDA Version | Stable Image Tag | Development Image Tag | Note |
|
||||||
|--------------|-----------------------------------|-----------------------------------|----------------------------------------------------------------------|
|
|--------------|-----------------------------------|-----------------------------------|----------------------------------------------------------------------|
|
||||||
| 12.1.0 | `runpod/worker-v1-vllm:v15.0stable-cuda12.1.0` | `runpod/worker-v1-vllm:v1.5.0dev-cuda12.1.0` | When creating an Endpoint, select CUDA Version 12.3, 12.2 and 12.1 in the filter. |
|
| 12.1.0 | `runpod/worker-v1-vllm:v2.2.0stable-cuda12.1.0` | `runpod/worker-v1-vllm:v2.2.0dev-cuda12.1.0` | When creating an Endpoint, select CUDA Version 12.3, 12.2 and 12.1 in the filter. |
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
@@ -488,7 +488,15 @@ The prompt string can be any string, and the model's chat template will not be a
|
|||||||
|
|
||||||
Example:
|
Example:
|
||||||
```json
|
```json
|
||||||
"prompt": "..."
|
{
|
||||||
|
"input": {
|
||||||
|
"prompt": "why sky is blue?",
|
||||||
|
"sampling_params": {
|
||||||
|
"temperature": 0.7,
|
||||||
|
"max_tokens": 100
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
```
|
```
|
||||||
2. `messages`
|
2. `messages`
|
||||||
Your list can contain any number of messages, and each message usually can have any role from the following list:
|
Your list can contain any number of messages, and each message usually can have any role from the following list:
|
||||||
@@ -502,20 +510,28 @@ Your list can contain any number of messages, and each message usually can have
|
|||||||
|
|
||||||
Example:
|
Example:
|
||||||
```json
|
```json
|
||||||
"messages": [
|
{
|
||||||
{
|
"input": {
|
||||||
"role": "system",
|
"messages": [
|
||||||
"content": "..."
|
{
|
||||||
},
|
"role": "system",
|
||||||
{
|
"content": "You are a helpful AI assistant that provides clear and concise responses."
|
||||||
"role": "user",
|
},
|
||||||
"content": "..."
|
{
|
||||||
},
|
"role": "user",
|
||||||
{
|
"content": "Can you explain the difference between supervised and unsupervised learning?"
|
||||||
"role": "assistant",
|
},
|
||||||
"content": "..."
|
{
|
||||||
|
"role": "assistant",
|
||||||
|
"content": "Sure! Supervised learning uses labeled data, meaning each input has a corresponding correct output. The model learns by mapping inputs to known outputs. In contrast, unsupervised learning works with unlabeled data, where the model identifies patterns, structures, or clusters without predefined answers."
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"sampling_params": {
|
||||||
|
"temperature": 0.7,
|
||||||
|
"max_tokens": 100
|
||||||
}
|
}
|
||||||
]
|
}
|
||||||
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
</details>
|
</details>
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
ray
|
ray
|
||||||
pandas
|
pandas
|
||||||
pyarrow
|
pyarrow
|
||||||
runpod==1.7.1
|
runpod~=1.7.7
|
||||||
huggingface-hub
|
huggingface-hub
|
||||||
packaging
|
packaging
|
||||||
typing-extensions==4.7.1
|
typing-extensions==4.7.1
|
||||||
|
|||||||
+1
-1
@@ -7,7 +7,7 @@ variable "REPOSITORY" {
|
|||||||
}
|
}
|
||||||
|
|
||||||
variable "BASE_IMAGE_VERSION" {
|
variable "BASE_IMAGE_VERSION" {
|
||||||
default = "stable"
|
default = "v2.0.0stable"
|
||||||
}
|
}
|
||||||
|
|
||||||
group "all" {
|
group "all" {
|
||||||
|
|||||||
+34
-14
@@ -4,14 +4,16 @@ import json
|
|||||||
import asyncio
|
import asyncio
|
||||||
|
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
from typing import AsyncGenerator
|
from typing import AsyncGenerator, Optional
|
||||||
import time
|
import time
|
||||||
|
|
||||||
from vllm import AsyncLLMEngine
|
from vllm import AsyncLLMEngine
|
||||||
|
from vllm.entrypoints.logger import RequestLogger
|
||||||
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
|
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
|
||||||
from vllm.entrypoints.openai.serving_completion import OpenAIServingCompletion
|
from vllm.entrypoints.openai.serving_completion import OpenAIServingCompletion
|
||||||
from vllm.entrypoints.openai.protocol import ChatCompletionRequest, CompletionRequest, ErrorResponse
|
from vllm.entrypoints.openai.protocol import ChatCompletionRequest, CompletionRequest, ErrorResponse
|
||||||
from vllm.entrypoints.openai.serving_engine import BaseModelPath
|
from vllm.entrypoints.openai.serving_models import BaseModelPath, LoRAModulePath, OpenAIServingModels
|
||||||
|
|
||||||
|
|
||||||
from utils import DummyRequest, JobInput, BatchSize, create_error_response
|
from utils import DummyRequest, JobInput, BatchSize, create_error_response
|
||||||
from constants import DEFAULT_MAX_CONCURRENCY, DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MIN_BATCH_SIZE
|
from constants import DEFAULT_MAX_CONCURRENCY, DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MIN_BATCH_SIZE
|
||||||
@@ -128,23 +130,44 @@ class OpenAIvLLMEngine(vLLMEngine):
|
|||||||
self.base_model_paths = [
|
self.base_model_paths = [
|
||||||
BaseModelPath(name=self.engine_args.model, model_path=self.engine_args.model)
|
BaseModelPath(name=self.engine_args.model, model_path=self.engine_args.model)
|
||||||
]
|
]
|
||||||
self.chat_engine = OpenAIServingChat(
|
|
||||||
|
lora_modules = os.getenv('LORA_MODULES', None)
|
||||||
|
if lora_modules is not None:
|
||||||
|
try:
|
||||||
|
lora_modules = json.loads(lora_modules)
|
||||||
|
lora_modules = [LoRAModulePath(**lora_modules)]
|
||||||
|
except:
|
||||||
|
lora_modules = None
|
||||||
|
|
||||||
|
self.serving_models = OpenAIServingModels(
|
||||||
engine_client=self.llm,
|
engine_client=self.llm,
|
||||||
model_config=self.model_config,
|
model_config=self.model_config,
|
||||||
base_model_paths=self.base_model_paths,
|
base_model_paths=self.base_model_paths,
|
||||||
response_role=self.response_role,
|
|
||||||
chat_template=self.tokenizer.tokenizer.chat_template,
|
|
||||||
lora_modules=None,
|
lora_modules=None,
|
||||||
prompt_adapters=None,
|
prompt_adapters=None,
|
||||||
request_logger=None
|
)
|
||||||
|
|
||||||
|
self.chat_engine = OpenAIServingChat(
|
||||||
|
engine_client=self.llm,
|
||||||
|
model_config=self.model_config,
|
||||||
|
models=self.serving_models,
|
||||||
|
response_role=self.response_role,
|
||||||
|
request_logger=None,
|
||||||
|
chat_template=self.tokenizer.tokenizer.chat_template,
|
||||||
|
chat_template_content_format="auto",
|
||||||
|
# enable_reasoning=os.getenv('ENABLE_REASONING', 'false').lower() == 'true',
|
||||||
|
# reasoning_parser=None,
|
||||||
|
# return_token_as_token_ids=False,
|
||||||
|
enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true',
|
||||||
|
tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None,
|
||||||
|
enable_prompt_tokens_details=False
|
||||||
)
|
)
|
||||||
self.completion_engine = OpenAIServingCompletion(
|
self.completion_engine = OpenAIServingCompletion(
|
||||||
engine_client=self.llm,
|
engine_client=self.llm,
|
||||||
model_config=self.model_config,
|
model_config=self.model_config,
|
||||||
base_model_paths=self.base_model_paths,
|
models=self.serving_models,
|
||||||
lora_modules=[],
|
request_logger=None,
|
||||||
prompt_adapters=None,
|
# return_token_as_token_ids=False,
|
||||||
request_logger=None
|
|
||||||
)
|
)
|
||||||
|
|
||||||
async def generate(self, openai_request: JobInput):
|
async def generate(self, openai_request: JobInput):
|
||||||
@@ -157,10 +180,7 @@ class OpenAIvLLMEngine(vLLMEngine):
|
|||||||
yield create_error_response("Invalid route").model_dump()
|
yield create_error_response("Invalid route").model_dump()
|
||||||
|
|
||||||
async def _handle_model_request(self):
|
async def _handle_model_request(self):
|
||||||
models = await self.chat_engine.show_available_models()
|
models = await self.serving_models.show_available_models()
|
||||||
fixed_model = models.data[0]
|
|
||||||
fixed_model.id = self.served_model_name
|
|
||||||
models.data = [fixed_model]
|
|
||||||
return models.model_dump()
|
return models.model_dump()
|
||||||
|
|
||||||
async def _handle_chat_or_completion_request(self, openai_request: JobInput):
|
async def _handle_chat_or_completion_request(self, openai_request: JobInput):
|
||||||
|
|||||||
+3
-1
@@ -4,6 +4,7 @@ import logging
|
|||||||
from torch.cuda import device_count
|
from torch.cuda import device_count
|
||||||
from vllm import AsyncEngineArgs
|
from vllm import AsyncEngineArgs
|
||||||
from vllm.model_executor.model_loader.tensorizer import TensorizerConfig
|
from vllm.model_executor.model_loader.tensorizer import TensorizerConfig
|
||||||
|
from src.utils import convert_limit_mm_per_prompt
|
||||||
|
|
||||||
RENAME_ARGS_MAP = {
|
RENAME_ARGS_MAP = {
|
||||||
"MODEL_NAME": "model",
|
"MODEL_NAME": "model",
|
||||||
@@ -88,7 +89,8 @@ DEFAULT_ARGS = {
|
|||||||
"typical_acceptance_sampler_posterior_alpha": float(os.getenv('TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA', 0)) or None,
|
"typical_acceptance_sampler_posterior_alpha": float(os.getenv('TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA', 0)) or None,
|
||||||
"qlora_adapter_name_or_path": os.getenv('QLORA_ADAPTER_NAME_OR_PATH', None),
|
"qlora_adapter_name_or_path": os.getenv('QLORA_ADAPTER_NAME_OR_PATH', None),
|
||||||
"disable_logprobs_during_spec_decoding": os.getenv('DISABLE_LOGPROBS_DURING_SPEC_DECODING', None),
|
"disable_logprobs_during_spec_decoding": os.getenv('DISABLE_LOGPROBS_DURING_SPEC_DECODING', None),
|
||||||
"otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None)
|
"otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None),
|
||||||
|
"use_v2_block_manager": os.getenv('USE_V2_BLOCK_MANAGER', 'true'),
|
||||||
}
|
}
|
||||||
|
|
||||||
def match_vllm_args(args):
|
def match_vllm_args(args):
|
||||||
|
|||||||
+9
-1
@@ -15,6 +15,10 @@ except ImportError:
|
|||||||
|
|
||||||
logging.basicConfig(level=logging.INFO)
|
logging.basicConfig(level=logging.INFO)
|
||||||
|
|
||||||
|
def convert_limit_mm_per_prompt(input_string: str):
|
||||||
|
key, value = input_string.split('=')
|
||||||
|
return {key: int(value)}
|
||||||
|
|
||||||
def count_physical_cores():
|
def count_physical_cores():
|
||||||
with open('/proc/cpuinfo') as f:
|
with open('/proc/cpuinfo') as f:
|
||||||
content = f.readlines()
|
content = f.readlines()
|
||||||
@@ -40,7 +44,11 @@ class JobInput:
|
|||||||
self.max_batch_size = job.get("max_batch_size")
|
self.max_batch_size = job.get("max_batch_size")
|
||||||
self.apply_chat_template = job.get("apply_chat_template", False)
|
self.apply_chat_template = job.get("apply_chat_template", False)
|
||||||
self.use_openai_format = job.get("use_openai_format", False)
|
self.use_openai_format = job.get("use_openai_format", False)
|
||||||
self.sampling_params = SamplingParams(**job.get("sampling_params", {}))
|
samp_param = job.get("sampling_params", {})
|
||||||
|
if "max_tokens" not in samp_param:
|
||||||
|
samp_param["max_tokens"] = 100
|
||||||
|
self.sampling_params = SamplingParams(**samp_param)
|
||||||
|
# self.sampling_params = SamplingParams(max_tokens=100, **job.get("sampling_params", {}))
|
||||||
self.request_id = random_uuid()
|
self.request_id = random_uuid()
|
||||||
batch_size_growth_factor = job.get("batch_size_growth_factor")
|
batch_size_growth_factor = job.get("batch_size_growth_factor")
|
||||||
self.batch_size_growth_factor = float(batch_size_growth_factor) if batch_size_growth_factor else None
|
self.batch_size_growth_factor = float(batch_size_growth_factor) if batch_size_growth_factor else None
|
||||||
|
|||||||
+1165
-895
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user