Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9b7ca4d0b0 | ||
|
|
2a4eaf0356 | ||
|
|
ba19cc97bf | ||
|
|
6075f2c590 | ||
|
|
a9786a2481 | ||
|
|
0b6bc7a2be | ||
|
|
d0ab58ee17 | ||
|
|
4e474c41c8 | ||
|
|
e9d13c155e | ||
|
|
e807342e90 | ||
|
|
f33e8d2bcd | ||
|
|
6b64bb93bd | ||
|
|
a53cf777ff | ||
|
|
cfc258674b | ||
|
|
8beafed06b | ||
|
|
d77c53c3b7 | ||
|
|
3d067cd472 | ||
|
|
f1360ccae7 | ||
|
|
96f1b86126 | ||
|
|
3a095f3e10 | ||
|
|
7cf7f3e4e6 | ||
|
|
4395d0c67b | ||
|
|
5864fa6843 | ||
|
|
067f0bc173 | ||
|
|
0869068e7c | ||
|
|
19f25b17de | ||
|
|
70aa748c30 | ||
|
|
da3e5f524b | ||
|
|
acbdf63de8 | ||
|
|
fa31d9663d | ||
|
|
b8fb313483 | ||
|
|
9635daf336 | ||
|
|
fb4e700090 | ||
|
|
d4872ed9c8 | ||
|
|
b2eef50c3b | ||
|
|
8005bcc1a8 | ||
|
|
aa7b00ddda | ||
|
|
dc6f3239bd | ||
|
|
f9d0fcb78c | ||
|
|
99b952e55e | ||
|
|
389fad7952 | ||
|
|
56dc4ad075 | ||
|
|
6dcf39e159 | ||
|
|
2b1d618287 | ||
|
|
d7e9c49fe4 | ||
|
|
c9791f1163 | ||
|
|
6fc770415d | ||
|
|
04288240f6 | ||
|
|
9e8d9196b0 | ||
|
|
30dd7c1eb5 | ||
|
|
aa7c37ffb0 | ||
|
|
dc8c88027a | ||
|
|
a948e90caa | ||
|
|
c703254f71 | ||
|
|
2747106403 | ||
|
|
91ed30e9a2 | ||
|
|
fcbfe84f63 | ||
|
|
610429df23 | ||
|
|
a578c6df23 | ||
|
|
9299f43b8b | ||
|
|
131c17569f | ||
|
|
8882f6d50b | ||
|
|
331bc30101 | ||
|
|
a27f72a33a | ||
|
|
7167985f23 | ||
|
|
0e3359a70e | ||
|
|
3f0a20d28e | ||
|
|
8e3c26be14 | ||
|
|
66ea8b1110 | ||
|
|
a3d432afdf | ||
|
|
06c2bb1715 | ||
|
|
d2e355eae9 | ||
|
|
b7787d8ca8 | ||
|
|
b1fca5d257 | ||
|
|
3e86d16892 | ||
|
|
d9f54ce76a | ||
|
|
149da95cd0 | ||
|
|
0a89394f1d | ||
|
|
8df7f41f1d | ||
|
|
4d7b8c03c0 | ||
|
|
6c6bf50379 | ||
|
|
27a2ee5754 | ||
|
|
2df915a145 | ||
|
|
aadc025849 | ||
|
|
8b4a49073d | ||
|
|
4e10641d69 | ||
|
|
6e8696c12a | ||
|
|
b49e81a75a | ||
|
|
65932f85e1 | ||
|
|
94840cfbbb | ||
|
|
ce47c41f4a | ||
|
|
c03ecc42fe | ||
|
|
ae56b9f43d | ||
|
|
891699be1e | ||
|
|
c28aa02576 | ||
|
|
eba20c0704 | ||
|
|
677a01e8f3 | ||
|
|
5cd12ba331 | ||
|
|
850c686538 | ||
|
|
de2876e659 | ||
|
|
d3ee3236c0 | ||
|
|
0781e93054 | ||
|
|
251e807012 | ||
|
|
29346769ed | ||
|
|
1420091588 | ||
|
|
d46adeaee9 | ||
|
|
5c0dca44bd | ||
|
|
cc301ac123 | ||
|
|
2ac6a0108f | ||
|
|
b1554ea10f | ||
|
|
0d794a3914 | ||
|
|
3dad3a754c | ||
|
|
44cef385df | ||
|
|
814f50af38 | ||
|
|
f3a530b7fe | ||
|
|
cdf78e7c89 | ||
|
|
ab40d9c9a8 | ||
|
|
3293245c81 | ||
|
|
5e1c8c8128 | ||
|
|
286d6ba702 | ||
|
|
8d734f8340 | ||
|
|
4fa4a8e0e6 | ||
|
|
39ce8a64c0 | ||
|
|
825ef25b60 | ||
|
|
1e9aeb6e8f | ||
|
|
e6172dddd4 | ||
|
|
21a1e138b4 | ||
|
|
a40e7803ee | ||
|
|
5e245793bc | ||
|
|
2111c9e7a5 | ||
|
|
0ae11ea6df | ||
|
|
7f46582949 | ||
|
|
571ef2b805 | ||
|
|
967eaba573 | ||
|
|
eb75a3ac03 | ||
|
|
9cb9336cf5 | ||
|
|
6a15a9e750 | ||
|
|
f023f57217 | ||
|
|
673597fd46 | ||
|
|
c50543ebd9 | ||
|
|
17a2d844ec | ||
|
|
3498e99b2f | ||
|
|
5da96ce9a6 | ||
|
|
e32626ca9d |
@@ -19,32 +19,49 @@ jobs:
|
||||
|
||||
- name: Check for new package version and update
|
||||
run: |
|
||||
# Get current version
|
||||
current_version=$(grep -oP 'runpod==\K[^"]+' ./builder/requirements.txt)
|
||||
echo "Fetching the current runpod version from requirements.txt..."
|
||||
|
||||
# Get new version
|
||||
# Get current version, allowing both == and ~= in the search pattern
|
||||
current_version=$(grep -oP 'runpod[~=]{1,2}\K[^"]+' ./builder/requirements.txt)
|
||||
echo "Current version: $current_version"
|
||||
|
||||
# Extract major and minor from current version
|
||||
current_major_minor=$(echo $current_version | cut -d. -f1,2)
|
||||
echo "Current major.minor: $current_major_minor"
|
||||
|
||||
echo "Fetching the latest runpod version from PyPI..."
|
||||
|
||||
# Get new version from PyPI
|
||||
new_version=$(curl -s https://pypi.org/pypi/runpod/json | jq -r .info.version)
|
||||
echo "NEW_VERSION_ENV=$new_version" >> $GITHUB_ENV
|
||||
echo "New version: $new_version"
|
||||
|
||||
# Extract major and minor from new version
|
||||
new_major_minor=$(echo $new_version | cut -d. -f1,2)
|
||||
echo "New major.minor: $new_major_minor"
|
||||
|
||||
if [ -z "$new_version" ]; then
|
||||
echo "Failed to fetch the new version."
|
||||
echo "ERROR: Failed to fetch the new version from PyPI."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Check if the version is already up-to-date
|
||||
if [ "$current_version" = "$new_version" ]; then
|
||||
echo "The package version is already up-to-date."
|
||||
# Check if the major or minor version is different
|
||||
if [ "$current_major_minor" = "$new_major_minor" ]; then
|
||||
echo "No update needed. The new version ($new_major_minor) is within the allowed range (~= $current_major_minor)."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Update requirements.txt
|
||||
sed -i "s/runpod==.*/runpod==$new_version/" ./builder/requirements.txt
|
||||
echo "New major/minor detected ($new_major_minor). Updating requirements.txt..."
|
||||
|
||||
# Update requirements.txt, preserving the existing constraint type (~= or ==)
|
||||
sed -i "s/runpod[~=][^ ]*/runpod~=$new_version/" ./builder/requirements.txt
|
||||
echo "requirements.txt has been updated."
|
||||
|
||||
- name: Create Pull Request
|
||||
uses: peter-evans/create-pull-request@v3
|
||||
with:
|
||||
token: ${{ secrets.GITHUB_TOKEN }}
|
||||
commit-message: Update package version
|
||||
commit-message: Update runpod package version
|
||||
title: Update runpod package version
|
||||
body: The package version has been updated to ${{ env.NEW_VERSION_ENV }}
|
||||
branch: runpod-package-update
|
||||
|
||||
+1016
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,32 @@
|
||||
{
|
||||
"tests": [
|
||||
{
|
||||
"name": "basic_inference_test",
|
||||
"input": {
|
||||
"prompt": "Write a short poem about artificial intelligence."
|
||||
},
|
||||
"timeout": 30000
|
||||
}
|
||||
],
|
||||
"config": {
|
||||
"gpuTypeId": "NVIDIA GeForce RTX 4090",
|
||||
"gpuCount": 1,
|
||||
"env": [
|
||||
{
|
||||
"key": "MODEL_NAME",
|
||||
"value": "facebook/opt-350m"
|
||||
}
|
||||
],
|
||||
"allowedCudaVersions": [
|
||||
"12.7",
|
||||
"12.6",
|
||||
"12.5",
|
||||
"12.4",
|
||||
"12.3",
|
||||
"12.2",
|
||||
"12.1",
|
||||
"12.0",
|
||||
"11.7"
|
||||
]
|
||||
}
|
||||
}
|
||||
+2
-2
@@ -12,7 +12,7 @@ RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
python3 -m pip install --upgrade -r /requirements.txt
|
||||
|
||||
# Install vLLM (switching back to pip installs since issues that required building fork are fixed and space optimization is not as important since caching) and FlashInfer
|
||||
RUN python3 -m pip install vllm==0.5.3.post1 && \
|
||||
RUN python3 -m pip install vllm==0.8.4 && \
|
||||
python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3
|
||||
|
||||
# Setup for Option 2: Building the Image with the Model included
|
||||
@@ -32,7 +32,7 @@ ENV MODEL_NAME=$MODEL_NAME \
|
||||
HF_DATASETS_CACHE="${BASE_PATH}/huggingface-cache/datasets" \
|
||||
HUGGINGFACE_HUB_CACHE="${BASE_PATH}/huggingface-cache/hub" \
|
||||
HF_HOME="${BASE_PATH}/huggingface-cache/hub" \
|
||||
HF_HUB_ENABLE_HF_TRANSFER=1
|
||||
HF_HUB_ENABLE_HF_TRANSFER=0
|
||||
|
||||
ENV PYTHONPATH="/:/vllm-workspace"
|
||||
|
||||
|
||||
@@ -18,8 +18,9 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
|
||||
### 1. UI for Deploying vLLM Worker on RunPod console:
|
||||

|
||||
|
||||
### 2. Worker vLLM `v1.1` with vLLM `0.5.3` now available under `stable` tags
|
||||
Update v1.1 is now available, use the image tag `runpod/worker-v1-vllm:stable-cuda12.1.0`.
|
||||
### 2. Worker vLLM `v2.5.0` with vLLM `0.8.5` now available under `stable` tags
|
||||
|
||||
Update v2.5.0 is now available, use the image tag `runpod/worker-v1-vllm:v2.5.0stable-cuda12.1.0`.
|
||||
|
||||
### 3. OpenAI-Compatible [Embedding Worker](https://github.com/runpod-workers/worker-infinity-embedding) Released
|
||||
Deploy your own OpenAI-compatible Serverless Endpoint on RunPod with multiple embedding models and fast inference for RAG and more!
|
||||
@@ -37,9 +38,8 @@ Worker vLLM is now cached on all RunPod machines, resulting in near-instant depl
|
||||
- [Environment Variables](#environment-variables)
|
||||
- [LLM Settings](#llm-settings)
|
||||
- [Tokenizer Settings](#tokenizer-settings)
|
||||
- [Tensor Parallelism (Multi-GPU) Settings](#tensor-parallelism-multi-gpu-settings)
|
||||
- [System Settings](#system-settings)
|
||||
- [Streaming Batch Size](#streaming-batch-size)
|
||||
- [System and Parallelism Settings](#system-and-parallelism-settings)
|
||||
- [Streaming Batch Size Settings](#streaming-batch-size-settings)
|
||||
- [OpenAI Settings](#openai-settings)
|
||||
- [Serverless Settings](#serverless-settings)
|
||||
- [Option 2: Build Docker Image with Model Inside](#option-2-build-docker-image-with-model-inside)
|
||||
@@ -57,6 +57,10 @@ Worker vLLM is now cached on all RunPod machines, resulting in near-instant depl
|
||||
- [Input Request Parameters](#input-request-parameters)
|
||||
- [Text Input Formats](#text-input-formats)
|
||||
- [Sampling Parameters](#sampling-parameters)
|
||||
- [Worker Config](#worker-config)
|
||||
- [Writing your worker-config.json](#writing-your-worker-configjson)
|
||||
- [Example of schema](#example-of-schema)
|
||||
- [Example of versions](#example-of-versions)
|
||||
|
||||
# Setting up the Serverless Worker
|
||||
|
||||
@@ -77,7 +81,7 @@ Below is a summary of the available RunPod Worker images, categorized by image s
|
||||
|
||||
| CUDA Version | Stable Image Tag | Development Image Tag | Note |
|
||||
|--------------|-----------------------------------|-----------------------------------|----------------------------------------------------------------------|
|
||||
| 12.1.0 | `runpod/worker-v1-vllm:stable-cuda12.1.0` | `runpod/worker-v1-vllm:dev-cuda12.1.0` | When creating an Endpoint, select CUDA Version 12.3, 12.2 and 12.1 in the filter. |
|
||||
| 12.1.0 | `runpod/worker-v1-vllm:v2.5.0stable-cuda12.1.0` | `runpod/worker-v1-vllm:v2.5.0dev-cuda12.1.0` | When creating an Endpoint, select CUDA Version 12.3, 12.2 and 12.1 in the filter. |
|
||||
|
||||
|
||||
|
||||
@@ -86,20 +90,22 @@ Below is a summary of the available RunPod Worker images, categorized by image s
|
||||
#### Prerequisites
|
||||
- RunPod Account
|
||||
|
||||
#### Environment Variables/Settings
|
||||
> Note: `0` is equivalent to `False` and `1` is equivalent to `True` for boolean values.
|
||||
#### Environment Variables
|
||||
> Note: `0` is equivalent to `False` and `1` is equivalent to `True` for boolean as int values.
|
||||
|
||||
#### LLM Settings
|
||||
| `Name` | `Default` | `Type/Choices` | `Description` |
|
||||
|-------------------------------------------|-----------------------|--------------------------------------------|---------------|
|
||||
| `MODEL_NAME` | 'facebook/opt-125m' | `str` | Name or path of the Hugging Face model to use. |
|
||||
| `TOKENIZER` | None | `str` | Name or path of the Hugging Face tokenizer to use. |
|
||||
| `SKIP_TOKENIZER_INIT` | False | `bool` | Skip initialization of tokenizer and detokenizer. |
|
||||
| `TOKENIZER_MODE` | 'auto' | ['auto', 'slow'] | The tokenizer mode. |
|
||||
| `TRUST_REMOTE_CODE` | False | `bool` | Trust remote code from Hugging Face. |
|
||||
| `TRUST_REMOTE_CODE` | `False` | `bool` | Trust remote code from Hugging Face. |
|
||||
| `DOWNLOAD_DIR` | None | `str` | Directory to download and load the weights. |
|
||||
| `LOAD_FORMAT` | 'auto' | ['auto', 'pt', 'safetensors', 'npcache', 'dummy', 'tensorizer', 'bitsandbytes'] | The format of the model weights to load. |
|
||||
| `LOAD_FORMAT` | 'auto' | `str` | The format of the model weights to load. |
|
||||
| `HF_TOKEN` | - | `str` | Hugging Face token for private and gated models.|
|
||||
| `DTYPE` | 'auto' | ['auto', 'half', 'float16', 'bfloat16', 'float', 'float32'] | Data type for model weights and activations. |
|
||||
| `KV_CACHE_DTYPE` | 'auto' | ['auto', 'fp8', 'fp8_e5m2', 'fp8_e4m3'] | Data type for KV cache storage. |
|
||||
| `KV_CACHE_DTYPE` | 'auto' | ['auto', 'fp8'] | Data type for KV cache storage. |
|
||||
| `QUANTIZATION_PARAM_PATH` | None | `str` | Path to the JSON file containing the KV cache scaling factors. |
|
||||
| `MAX_MODEL_LEN` | None | `int` | Model context length. |
|
||||
| `GUIDED_DECODING_BACKEND` | 'outlines' | ['outlines', 'lm-format-enforcer'] | Which engine will be used for guided decoding by default. |
|
||||
@@ -109,26 +115,19 @@ Below is a summary of the available RunPod Worker images, categorized by image s
|
||||
| `TENSOR_PARALLEL_SIZE` | 1 | `int` | Number of tensor parallel replicas. |
|
||||
| `MAX_PARALLEL_LOADING_WORKERS` | None | `int` | Load model sequentially in multiple batches. |
|
||||
| `RAY_WORKERS_USE_NSIGHT` | False | `bool` | If specified, use nsight to profile Ray workers. |
|
||||
| `BLOCK_SIZE` | 16 | [8, 16, 32] | Token block size for contiguous chunks of tokens. |
|
||||
| `ENABLE_PREFIX_CACHING` | False | `bool` | Enables automatic prefix caching. |
|
||||
| `DISABLE_SLIDING_WINDOW` | False | `bool` | Disables sliding window, capping to sliding window size. |
|
||||
| `USE_V2_BLOCK_MANAGER` | False | `bool` | Use BlockSpaceMangerV2. |
|
||||
| `NUM_LOOKAHEAD_SLOTS` | 0 | `int` | Experimental scheduling config necessary for speculative decoding. |
|
||||
| `SEED` | 0 | `int` | Random seed for operations. |
|
||||
| `SWAP_SPACE` | 4 | `int` | CPU swap space size (GiB) per GPU. |
|
||||
| `GPU_MEMORY_UTILIZATION` | 0.90 | `float` | The fraction of GPU memory to be used for the model executor. |
|
||||
| `NUM_GPU_BLOCKS_OVERRIDE` | None | `int` | If specified, ignore GPU profiling result and use this number of GPU blocks. |
|
||||
| `MAX_NUM_BATCHED_TOKENS` | None | `int` | Maximum number of batched tokens per iteration. |
|
||||
| `MAX_NUM_SEQS` | 256 | `int` | Maximum number of sequences per iteration. |
|
||||
| `MAX_LOGPROBS` | 20 | `int` | Max number of log probs to return when logprobs is specified in SamplingParams. |
|
||||
| `DISABLE_LOG_STATS` | False | `bool` | Disable logging statistics. |
|
||||
| `QUANTIZATION` | None | [*QUANTIZATION_METHODS, None] | Method used to quantize the weights. |
|
||||
| `QUANTIZATION` | None | ['awq', 'squeezellm', 'gptq', 'bitsandbytes'] | Method used to quantize the weights. |
|
||||
| `ROPE_SCALING` | None | `dict` | RoPE scaling configuration in JSON format. |
|
||||
| `ROPE_THETA` | None | `float` | RoPE theta. Use with rope_scaling. |
|
||||
| `ENFORCE_EAGER` | False | `bool` | Always use eager-mode PyTorch. |
|
||||
| `MAX_CONTEXT_LEN_TO_CAPTURE` | None | `int` | Maximum context length covered by CUDA graphs. |
|
||||
| `MAX_SEQ_LEN_TO_CAPTURE` | 8192 | `int` | Maximum sequence length covered by CUDA graphs. |
|
||||
| `DISABLE_CUSTOM_ALL_REDUCE` | False | `bool` | See ParallelConfig. |
|
||||
| `TOKENIZER_POOL_SIZE` | 0 | `int` | Size of tokenizer pool to use for asynchronous tokenization. |
|
||||
| `TOKENIZER_POOL_TYPE` | 'ray' | `str` | Type of tokenizer pool to use for asynchronous tokenization. |
|
||||
| `TOKENIZER_POOL_EXTRA_CONFIG` | None | `dict` | Extra config for tokenizer pool. |
|
||||
@@ -140,7 +139,6 @@ Below is a summary of the available RunPod Worker images, categorized by image s
|
||||
| `LONG_LORA_SCALING_FACTORS` | None | `tuple` | Specify multiple scaling factors for LoRA adapters. |
|
||||
| `MAX_CPU_LORAS` | None | `int` | Maximum number of LoRAs to store in CPU memory. |
|
||||
| `FULLY_SHARDED_LORAS` | False | `bool` | Enable fully sharded LoRA layers. |
|
||||
| `DEVICE` | 'auto' | ['auto', 'cuda', 'neuron', 'cpu', 'openvino', 'tpu', 'xpu'] | Device type for vLLM execution. |
|
||||
| `SCHEDULER_DELAY_FACTOR` | 0.0 | `float` | Apply a delay before scheduling next prompt. |
|
||||
| `ENABLE_CHUNKED_PREFILL` | False | `bool` | Enable chunked prefill requests. |
|
||||
| `SPECULATIVE_MODEL` | None | `str` | The name of the draft model to be used in speculative decoding. |
|
||||
@@ -159,31 +157,55 @@ Below is a summary of the available RunPod Worker images, categorized by image s
|
||||
| `PREEMPTION_CPU_CAPACITY` | 2 | `float` | The percentage of CPU memory used for the saved activations. |
|
||||
| `DISABLE_LOGGING_REQUEST` | False | `bool` | Disable logging requests. |
|
||||
| `MAX_LOG_LEN` | None | `int` | Max number of prompt characters or prompt ID numbers being printed in log. |
|
||||
**Tokenizer Settings**
|
||||
|
||||
|
||||
#### Tokenizer Settings
|
||||
|
||||
| `Name` | `Default` | `Type/Choices` | `Description` |
|
||||
|-------------------------------------------|-----------------------|--------------------------------------------|---------------|
|
||||
| `TOKENIZER_NAME` | `None` | `str` |Tokenizer repository to use a different tokenizer than the model's default. |
|
||||
| `TOKENIZER_REVISION` | `None` | `str` |Tokenizer revision to load. |
|
||||
| `CUSTOM_CHAT_TEMPLATE` | `None` | `str` of single-line jinja template |Custom chat jinja template. [More Info](https://huggingface.co/docs/transformers/chat_templating) |
|
||||
**System, GPU, and Tensor Parallelism(Multi-GPU) Settings**
|
||||
|
||||
#### System and Parallelism Settings
|
||||
|
||||
| `Name` | `Default` | `Type/Choices` | `Description` |
|
||||
|-------------------------------------------|-----------------------|--------------------------------------------|---------------|
|
||||
| `GPU_MEMORY_UTILIZATION` | `0.95` | `float` |Sets GPU VRAM utilization. |
|
||||
| `MAX_PARALLEL_LOADING_WORKERS` | `None` | `int` |Load model sequentially in multiple batches, to avoid RAM OOM when using tensor parallel and large models. |
|
||||
| `BLOCK_SIZE` | `16` | `8`, `16`, `32` |Token block size for contiguous chunks of tokens. |
|
||||
| `SWAP_SPACE` | `4` | `int` |CPU swap space size (GiB) per GPU. |
|
||||
| `ENFORCE_EAGER` | `0` | boolean as `int` |Always use eager-mode PyTorch. If False(`0`), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility. |
|
||||
| `ENFORCE_EAGER` | False | `bool` |Always use eager-mode PyTorch. If False(`0`), will use eager mode and CUDA graph in hybrid for maximal performance and flexibility. |
|
||||
| `MAX_SEQ_LEN_TO_CAPTURE` | `8192` | `int` |Maximum context length covered by CUDA graphs. When a sequence has context length larger than this, we fall back to eager mode.|
|
||||
| `DISABLE_CUSTOM_ALL_REDUCE` | `0` | `int` |Enables or disables custom all reduce. |
|
||||
**Streaming Batch Size Settings**:
|
||||
|
||||
|
||||
#### Streaming Batch Size Settings
|
||||
|
||||
The way this works is that the first request will have a batch size of `DEFAULT_MIN_BATCH_SIZE`, and each subsequent request will have a batch size of `previous_batch_size * DEFAULT_BATCH_SIZE_GROWTH_FACTOR`. This will continue until the batch size reaches `DEFAULT_BATCH_SIZE`. E.g. for the default values, the batch sizes will be `1, 3, 9, 27, 50, 50, 50, ...`. You can also specify this per request, with inputs `max_batch_size`, `min_batch_size`, and `batch_size_growth_factor`. This has nothing to do with vLLM's internal batching, but rather the number of tokens sent in each HTTP request from the worker
|
||||
|
||||
|
||||
| `Name` | `Default` | `Type/Choices` | `Description` |
|
||||
|-------------------------------------------|-----------------------|--------------------------------------------|---------------|
|
||||
| `DEFAULT_BATCH_SIZE` | `50` | `int` |Default and Maximum batch size for token streaming to reduce HTTP calls. |
|
||||
| `DEFAULT_MIN_BATCH_SIZE` | `1` | `int` |Batch size for the first request, which will be multiplied by the growth factor every subsequent request. |
|
||||
| `DEFAULT_BATCH_SIZE_GROWTH_FACTOR` | `3` | `float` |Growth factor for dynamic batch size. |
|
||||
The way this works is that the first request will have a batch size of `DEFAULT_MIN_BATCH_SIZE`, and each subsequent request will have a batch size of `previous_batch_size * DEFAULT_BATCH_SIZE_GROWTH_FACTOR`. This will continue until the batch size reaches `DEFAULT_BATCH_SIZE`. E.g. for the default values, the batch sizes will be `1, 3, 9, 27, 50, 50, 50, ...`. You can also specify this per request, with inputs `max_batch_size`, `min_batch_size`, and `batch_size_growth_factor`. This has nothing to do with vLLM's internal batching, but rather the number of tokens sent in each HTTP request from the worker |
|
||||
**OpenAI Settings**
|
||||
|
||||
#### OpenAI Settings
|
||||
|
||||
| `Name` | `Default` | `Type/Choices` | `Description` |
|
||||
|-------------------------------------------|-----------------------|--------------------------------------------|---------------|
|
||||
| `RAW_OPENAI_OUTPUT` | `1` | boolean as `int` |Enables raw OpenAI SSE format string output when streaming. **Required** to be enabled (which it is by default) for OpenAI compatibility. |
|
||||
| `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | `None` | `str` |Overrides the name of the served model from model repo/path to specified name, which you will then be able to use the value for the `model` parameter when making OpenAI requests |
|
||||
| `OPENAI_RESPONSE_ROLE` | `assistant` | `str` |Role of the LLM's Response in OpenAI Chat Completions. |
|
||||
**Serverless Settings**
|
||||
|
||||
#### Serverless Settings
|
||||
|
||||
| `Name` | `Default` | `Type/Choices` | `Description` |
|
||||
|-------------------------------------------|-----------------------|--------------------------------------------|---------------|
|
||||
| `MAX_CONCURRENCY` | `300` | `int` |Max concurrent requests per worker. vLLM has an internal queue, so you don't have to worry about limiting by VRAM, this is for improving scaling/load balancing efficiency |
|
||||
| `DISABLE_LOG_STATS` | `1` | boolean as `int` |Enables or disables vLLM stats logging. |
|
||||
| `DISABLE_LOG_REQUESTS` | `1` | boolean as `int` |Enables or disables vLLM request logging. |
|
||||
| `DISABLE_LOG_STATS` | False | `bool` |Enables or disables vLLM stats logging. |
|
||||
| `DISABLE_LOG_REQUESTS` | False | `bool` |Enables or disables vLLM request logging. |
|
||||
|
||||
> [!TIP]
|
||||
> If you are facing issues when using Mixtral 8x7B, Quantized models, or handling unusual models/architectures, try setting `TRUST_REMOTE_CODE` to `1`.
|
||||
@@ -490,7 +512,15 @@ The prompt string can be any string, and the model's chat template will not be a
|
||||
|
||||
Example:
|
||||
```json
|
||||
"prompt": "..."
|
||||
{
|
||||
"input": {
|
||||
"prompt": "why sky is blue?",
|
||||
"sampling_params": {
|
||||
"temperature": 0.7,
|
||||
"max_tokens": 100
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
2. `messages`
|
||||
Your list can contain any number of messages, and each message usually can have any role from the following list:
|
||||
@@ -504,19 +534,110 @@ Your list can contain any number of messages, and each message usually can have
|
||||
|
||||
Example:
|
||||
```json
|
||||
{
|
||||
"input": {
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": "..."
|
||||
"content": "You are a helpful AI assistant that provides clear and concise responses."
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "..."
|
||||
"content": "Can you explain the difference between supervised and unsupervised learning?"
|
||||
},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "..."
|
||||
"content": "Sure! Supervised learning uses labeled data, meaning each input has a corresponding correct output. The model learns by mapping inputs to known outputs. In contrast, unsupervised learning works with unlabeled data, where the model identifies patterns, structures, or clusters without predefined answers."
|
||||
}
|
||||
],
|
||||
"sampling_params": {
|
||||
"temperature": 0.7,
|
||||
"max_tokens": 100
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
```
|
||||
|
||||
</details>
|
||||
|
||||
# Worker Config
|
||||
The worker config is a JSON file that is used to build the form that helps users configure their serverless endpoint on the RunPod Web Interface.
|
||||
|
||||
Note: This is a new feature and only works for workers that use one model
|
||||
|
||||
## Writing your worker-config.json
|
||||
The JSON consists of two main parts, schema and versions.
|
||||
- `schema`: Here you specify the form fields that will be displayed to the user.
|
||||
- `env_var_name`: The name of the environment variable that is being set using the form field.
|
||||
- `value`: This is the default value of the form field. It will be shown in the UI as such unless the user changes it.
|
||||
- `title`: This is the title of the form field in the UI.
|
||||
- `description`: This is the description of the form field in the UI.
|
||||
- `required`: This is a boolean that specifies if the form field is required.
|
||||
- `type`: This is the type of the form field. Options are:
|
||||
- `text`: Environment variable is a string so user inputs text in form field.
|
||||
- `select`: User selects one option from the dropdown. You must provide the `options` key value pair after type if using this.
|
||||
- `toggle`: User toggles between true and false.
|
||||
- `number`: User inputs a number in the form field.
|
||||
- `options`: Specify the options the user can select from if the type is `select`. DO NOT include this unless the `type` is `select`.
|
||||
- `versions`: This is where you call the form fields specified in `schema` and organize them into categories.
|
||||
- `imageName`: This is the name of the Docker image that will be used to run the serverless endpoint.
|
||||
- `minimumCudaVersion`: This is the minimum CUDA version that is required to run the serverless endpoint.
|
||||
- `categories`: This is where you call the keys of the form fields specified in `schema` and organize them into categories. Each category is a toggle list of forms on the Web UI.
|
||||
- `title`: This is the title of the category in the UI.
|
||||
- `settings`: This is the array of settings schemas specified in `schema` associated with the category.
|
||||
|
||||
## Example of schema
|
||||
```json
|
||||
{
|
||||
"schema": {
|
||||
"TOKENIZER": {
|
||||
"env_var_name": "TOKENIZER",
|
||||
"value": "",
|
||||
"title": "Tokenizer",
|
||||
"description": "Name or path of the Hugging Face tokenizer to use.",
|
||||
"required": false,
|
||||
"type": "text"
|
||||
},
|
||||
"TOKENIZER_MODE": {
|
||||
"env_var_name": "TOKENIZER_MODE",
|
||||
"value": "auto",
|
||||
"title": "Tokenizer Mode",
|
||||
"description": "The tokenizer mode.",
|
||||
"required": false,
|
||||
"type": "select",
|
||||
"options": [
|
||||
{ "value": "auto", "label": "auto" },
|
||||
{ "value": "slow", "label": "slow" }
|
||||
]
|
||||
},
|
||||
...
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## Example of versions
|
||||
```json
|
||||
{
|
||||
"versions": {
|
||||
"0.5.4": {
|
||||
"imageName": "runpod/worker-v1-vllm:v1.2.0stable-cuda12.1.0",
|
||||
"minimumCudaVersion": "12.1",
|
||||
"categories": [
|
||||
{
|
||||
"title": "LLM Settings",
|
||||
"settings": [
|
||||
"TOKENIZER", "TOKENIZER_MODE", "OTHER_SETTINGS_SCHEMA_KEYS_YOU_HAVE_SPECIFIED_0", ...
|
||||
]
|
||||
},
|
||||
{
|
||||
"title": "Tokenizer Settings",
|
||||
"settings": [
|
||||
"OTHER_SETTINGS_SCHEMA_KEYS_0", "OTHER_SETTINGS_SCHEMA_KEYS_1", ...
|
||||
]
|
||||
},
|
||||
...
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
@@ -1,10 +1,12 @@
|
||||
ray
|
||||
pandas
|
||||
pyarrow
|
||||
runpod==1.6.2
|
||||
runpod~=1.7.7
|
||||
huggingface-hub
|
||||
packaging
|
||||
typing-extensions==4.7.1
|
||||
typing-extensions>=4.8.0
|
||||
pydantic
|
||||
pydantic-settings
|
||||
hf-transfer
|
||||
transformers
|
||||
bitsandbytes>=0.45.0
|
||||
|
||||
+1
-1
@@ -7,7 +7,7 @@ variable "REPOSITORY" {
|
||||
}
|
||||
|
||||
variable "BASE_IMAGE_VERSION" {
|
||||
default = "stable"
|
||||
default = "v2.4.0stable"
|
||||
}
|
||||
|
||||
group "all" {
|
||||
|
||||
+41
-17
@@ -4,13 +4,16 @@ import json
|
||||
import asyncio
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from typing import AsyncGenerator
|
||||
from typing import AsyncGenerator, Optional
|
||||
import time
|
||||
|
||||
from vllm import AsyncLLMEngine
|
||||
from vllm.entrypoints.logger import RequestLogger
|
||||
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
|
||||
from vllm.entrypoints.openai.serving_completion import OpenAIServingCompletion
|
||||
from vllm.entrypoints.openai.protocol import ChatCompletionRequest, CompletionRequest, ErrorResponse
|
||||
from vllm.entrypoints.openai.serving_models import BaseModelPath, LoRAModulePath, OpenAIServingModels
|
||||
|
||||
|
||||
from utils import DummyRequest, JobInput, BatchSize, create_error_response
|
||||
from constants import DEFAULT_MAX_CONCURRENCY, DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE_GROWTH_FACTOR, DEFAULT_MIN_BATCH_SIZE
|
||||
@@ -124,24 +127,47 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
|
||||
async def _initialize_engines(self):
|
||||
self.model_config = await self.llm.get_model_config()
|
||||
self.base_model_paths = [
|
||||
BaseModelPath(name=self.engine_args.model, model_path=self.engine_args.model)
|
||||
]
|
||||
|
||||
self.chat_engine = OpenAIServingChat(
|
||||
engine=self.llm,
|
||||
lora_modules = os.getenv('LORA_MODULES', None)
|
||||
if lora_modules is not None:
|
||||
try:
|
||||
lora_modules = json.loads(lora_modules)
|
||||
lora_modules = [LoRAModulePath(**lora_modules)]
|
||||
except:
|
||||
lora_modules = None
|
||||
|
||||
self.serving_models = OpenAIServingModels(
|
||||
engine_client=self.llm,
|
||||
model_config=self.model_config,
|
||||
served_model_names=[self.served_model_name],
|
||||
response_role=self.response_role,
|
||||
chat_template=self.tokenizer.tokenizer.chat_template,
|
||||
base_model_paths=self.base_model_paths,
|
||||
lora_modules=None,
|
||||
prompt_adapters=None,
|
||||
request_logger=None
|
||||
)
|
||||
|
||||
self.chat_engine = OpenAIServingChat(
|
||||
engine_client=self.llm,
|
||||
model_config=self.model_config,
|
||||
models=self.serving_models,
|
||||
response_role=self.response_role,
|
||||
request_logger=None,
|
||||
chat_template=self.tokenizer.tokenizer.chat_template,
|
||||
chat_template_content_format="auto",
|
||||
# enable_reasoning=os.getenv('ENABLE_REASONING', 'false').lower() == 'true',
|
||||
# reasoning_parser=None,
|
||||
# return_token_as_token_ids=False,
|
||||
enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true',
|
||||
tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None,
|
||||
enable_prompt_tokens_details=False
|
||||
)
|
||||
self.completion_engine = OpenAIServingCompletion(
|
||||
engine=self.llm,
|
||||
engine_client=self.llm,
|
||||
model_config=self.model_config,
|
||||
served_model_names=[self.served_model_name],
|
||||
lora_modules=[],
|
||||
prompt_adapters=None,
|
||||
request_logger=None
|
||||
models=self.serving_models,
|
||||
request_logger=None,
|
||||
# return_token_as_token_ids=False,
|
||||
)
|
||||
|
||||
async def generate(self, openai_request: JobInput):
|
||||
@@ -154,10 +180,7 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
yield create_error_response("Invalid route").model_dump()
|
||||
|
||||
async def _handle_model_request(self):
|
||||
models = await self.chat_engine.show_available_models()
|
||||
fixed_model = models.data[0]
|
||||
fixed_model.id = self.served_model_name
|
||||
models.data = [fixed_model]
|
||||
models = await self.serving_models.show_available_models()
|
||||
return models.model_dump()
|
||||
|
||||
async def _handle_chat_or_completion_request(self, openai_request: JobInput):
|
||||
@@ -176,7 +199,8 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
yield create_error_response(str(e)).model_dump()
|
||||
return
|
||||
|
||||
response_generator = await generator_function(request, raw_request=None)
|
||||
dummy_request = DummyRequest()
|
||||
response_generator = await generator_function(request, raw_request=dummy_request)
|
||||
|
||||
if not openai_request.openai_input.get("stream") or isinstance(response_generator, ErrorResponse):
|
||||
yield response_generator.model_dump()
|
||||
|
||||
+12
-7
@@ -4,6 +4,7 @@ import logging
|
||||
from torch.cuda import device_count
|
||||
from vllm import AsyncEngineArgs
|
||||
from vllm.model_executor.model_loader.tensorizer import TensorizerConfig
|
||||
from src.utils import convert_limit_mm_per_prompt
|
||||
|
||||
RENAME_ARGS_MAP = {
|
||||
"MODEL_NAME": "model",
|
||||
@@ -13,9 +14,9 @@ RENAME_ARGS_MAP = {
|
||||
}
|
||||
|
||||
DEFAULT_ARGS = {
|
||||
"disable_log_stats": True,
|
||||
"disable_log_requests": True,
|
||||
"gpu_memory_utilization": 0.9,
|
||||
"disable_log_stats": os.getenv('DISABLE_LOG_STATS', 'False').lower() == 'true',
|
||||
"disable_log_requests": os.getenv('DISABLE_LOG_REQUESTS', 'False').lower() == 'true',
|
||||
"gpu_memory_utilization": float(os.getenv('GPU_MEMORY_UTILIZATION', 0.95)),
|
||||
"pipeline_parallel_size": int(os.getenv('PIPELINE_PARALLEL_SIZE', 1)),
|
||||
"tensor_parallel_size": int(os.getenv('TENSOR_PARALLEL_SIZE', 1)),
|
||||
"served_model_name": os.getenv('SERVED_MODEL_NAME', None),
|
||||
@@ -88,7 +89,8 @@ DEFAULT_ARGS = {
|
||||
"typical_acceptance_sampler_posterior_alpha": float(os.getenv('TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA', 0)) or None,
|
||||
"qlora_adapter_name_or_path": os.getenv('QLORA_ADAPTER_NAME_OR_PATH', None),
|
||||
"disable_logprobs_during_spec_decoding": os.getenv('DISABLE_LOGPROBS_DURING_SPEC_DECODING', None),
|
||||
"otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None)
|
||||
"otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None),
|
||||
"use_v2_block_manager": os.getenv('USE_V2_BLOCK_MANAGER', 'true'),
|
||||
}
|
||||
|
||||
def match_vllm_args(args):
|
||||
@@ -146,6 +148,9 @@ def get_engine_args():
|
||||
# Rename and match to vllm args
|
||||
args = match_vllm_args(args)
|
||||
|
||||
if args.get("load_format") == "bitsandbytes":
|
||||
args["quantization"] = args["load_format"]
|
||||
|
||||
# Set tensor parallel size and max parallel loading workers if more than 1 GPU is available
|
||||
num_gpus = device_count()
|
||||
if num_gpus > 1:
|
||||
@@ -162,8 +167,8 @@ def get_engine_args():
|
||||
args["max_seq_len_to_capture"] = int(os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE"))
|
||||
logging.warning("Using MAX_CONTEXT_LEN_TO_CAPTURE is deprecated. Please use MAX_SEQ_LEN_TO_CAPTURE instead.")
|
||||
|
||||
if "gemma-2" in args.get("model", "").lower():
|
||||
os.environ["VLLM_ATTENTION_BACKEND"] = "FLASHINFER"
|
||||
logging.info("Using FLASHINFER for gemma-2 model.")
|
||||
# if "gemma-2" in args.get("model", "").lower():
|
||||
# os.environ["VLLM_ATTENTION_BACKEND"] = "FLASHINFER"
|
||||
# logging.info("Using FLASHINFER for gemma-2 model.")
|
||||
|
||||
return AsyncEngineArgs(**args)
|
||||
|
||||
+16
-1
@@ -3,6 +3,7 @@ import logging
|
||||
from http import HTTPStatus
|
||||
from functools import wraps
|
||||
from time import time
|
||||
from vllm.entrypoints.openai.protocol import RequestResponseMetadata
|
||||
|
||||
try:
|
||||
from vllm.utils import random_uuid
|
||||
@@ -14,6 +15,10 @@ except ImportError:
|
||||
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
|
||||
def convert_limit_mm_per_prompt(input_string: str):
|
||||
key, value = input_string.split('=')
|
||||
return {key: int(value)}
|
||||
|
||||
def count_physical_cores():
|
||||
with open('/proc/cpuinfo') as f:
|
||||
content = f.readlines()
|
||||
@@ -39,7 +44,11 @@ class JobInput:
|
||||
self.max_batch_size = job.get("max_batch_size")
|
||||
self.apply_chat_template = job.get("apply_chat_template", False)
|
||||
self.use_openai_format = job.get("use_openai_format", False)
|
||||
self.sampling_params = SamplingParams(**job.get("sampling_params", {}))
|
||||
samp_param = job.get("sampling_params", {})
|
||||
if "max_tokens" not in samp_param:
|
||||
samp_param["max_tokens"] = 100
|
||||
self.sampling_params = SamplingParams(**samp_param)
|
||||
# self.sampling_params = SamplingParams(max_tokens=100, **job.get("sampling_params", {}))
|
||||
self.request_id = random_uuid()
|
||||
batch_size_growth_factor = job.get("batch_size_growth_factor")
|
||||
self.batch_size_growth_factor = float(batch_size_growth_factor) if batch_size_growth_factor else None
|
||||
@@ -47,8 +56,14 @@ class JobInput:
|
||||
self.min_batch_size = int(min_batch_size) if min_batch_size else None
|
||||
self.openai_route = job.get("openai_route")
|
||||
self.openai_input = job.get("openai_input")
|
||||
class DummyState:
|
||||
def __init__(self):
|
||||
self.request_metadata = None
|
||||
|
||||
class DummyRequest:
|
||||
def __init__(self):
|
||||
self.headers = {}
|
||||
self.state = DummyState()
|
||||
async def is_disconnected(self):
|
||||
return False
|
||||
|
||||
|
||||
+1390
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user