Compare commits

...
17 Commits
Author SHA1 Message Date
chrisvelaandGitHub b7c6d4f9a2 feat: update dockerfile to 12.9.1 (#267)
Release / release (push) Waiting to run
* feat: update dockerfile to 12.9.1

* update readme on VLLM_NIGHTLY build arg
2026-02-19 10:13:14 +01:00
chrisvelaandGitHub d69cc021e8 Merge pull request #268 from runpod-workers/fix/spec-config-0-to-none
Release / release (push) Waiting to run
fix: spec config env vars should be none if zero
2026-02-18 15:51:51 -06:00
velaraptor-runpod 61faa8f137 fix: spec config env vars should be none if zero 2026-02-18 15:41:19 -06:00
chrisvelaandGitHub 1606cff557 Merge pull request #265 from runpod-workers/fix/zero-max-model-num_batches
Release / release (push) Waiting to run
fix: check for zero param and set to None
2026-02-13 15:26:06 -06:00
velaraptor-runpod e705c9494b fix: check for zero param and set to None 2026-02-13 15:23:54 -06:00
chrisvelaandGitHub b749aa5718 Merge pull request #264 from runpod-workers/fix/max_num_batched_tokens
Release / release (push) Waiting to run
fix: max num batched tokens
2026-02-13 12:38:01 -06:00
velaraptor-runpod 4705ba8a7c fix: check max_num_batched_tokenz if max_model_len not set 2026-02-13 03:29:52 -06:00
velaraptor-runpod 767c66c301 make minimal changes 2026-02-13 03:23:44 -06:00
velaraptor-runpod fefdbe21a9 update changes 2026-02-13 03:16:43 -06:00
velaraptor-runpod ee961ad28d Update hub.json 2026-02-13 03:08:19 -06:00
velaraptor-runpod 2e8c251447 Merge branch 'main' into feat/update-vllm-v0.15.0 2026-02-13 03:01:05 -06:00
velaraptor-runpod c3cf43b228 Update hub.json 2026-02-13 00:22:16 -06:00
velaraptor-runpod 7ec10b98cd Update utils.py 2026-02-12 15:28:31 -06:00
velaraptor-runpod 340bc0b3c6 fix: served model name 2026-02-10 21:42:58 -06:00
velaraptor-runpod e1e9ef74ad add changes from pr 2026-02-06 18:10:09 -06:00
velaraptor-runpod 461f89cea6 add torch-c-dlpack-ext requirement 2026-02-06 17:03:39 -06:00
velaraptor-runpod 8eb55b90c1 add changes for v0.15.0 2026-02-05 17:24:16 -06:00
7 changed files with 248 additions and 30 deletions
+59 -1
View File
@@ -187,6 +187,7 @@
"name": "Max Model Length", "name": "Max Model Length",
"type": "number", "type": "number",
"description": "Model context length.", "description": "Model context length.",
"default": null,
"advanced": true "advanced": true
} }
}, },
@@ -206,7 +207,8 @@
"value": "mp" "value": "mp"
} }
], ],
"advanced": true "advanced": true,
"default": "mp"
} }
}, },
{ {
@@ -293,6 +295,7 @@
"name": "Max Num Batched Tokens", "name": "Max Num Batched Tokens",
"type": "number", "type": "number",
"description": "Maximum number of batched tokens per iteration.", "description": "Maximum number of batched tokens per iteration.",
"default": null,
"advanced": true "advanced": true
} }
}, },
@@ -490,6 +493,61 @@
"advanced": true "advanced": true
} }
}, },
{
"key": "SPECULATIVE_CONFIG",
"input": {
"name": "Speculative Config (JSON)",
"type": "string",
"description": "Full speculative decoding configuration as a JSON string. Overrides individual speculative env vars.",
"advanced": true
}
},
{
"key": "SPECULATIVE_METHOD",
"input": {
"name": "Speculative Method",
"type": "string",
"description": "Speculative decoding method to use.",
"options": [
{ "label": "None", "value": "" },
{ "label": "Draft Model", "value": "draft_model" },
{ "label": "N-gram", "value": "ngram" },
{ "label": "EAGLE", "value": "eagle" },
{ "label": "EAGLE3", "value": "eagle3" },
{ "label": "Medusa", "value": "medusa" },
{ "label": "MLP Speculator", "value": "mlp_speculator" }
],
"default": "",
"advanced": true
}
},
{
"key": "SPECULATIVE_MODEL",
"input": {
"name": "Speculative Model",
"type": "string",
"description": "The name of the draft model to be used in speculative decoding.",
"advanced": true
}
},
{
"key": "NUM_SPECULATIVE_TOKENS",
"input": {
"name": "Num Speculative Tokens",
"type": "number",
"description": "The number of speculative tokens to sample from the draft model.",
"advanced": true
}
},
{
"key": "NGRAM_PROMPT_LOOKUP_MAX",
"input": {
"name": "Ngram Prompt Lookup Max",
"type": "number",
"description": "Max size of window for ngram prompt lookup in speculative decoding.",
"advanced": true
}
},
{ {
"key": "MODEL_LOADER_EXTRA_CONFIG", "key": "MODEL_LOADER_EXTRA_CONFIG",
"input": { "input": {
+9 -3
View File
@@ -1,13 +1,13 @@
FROM nvidia/cuda:12.8.0-base-ubuntu22.04 FROM nvidia/cuda:12.9.1-base-ubuntu22.04
RUN apt-get update -y \ RUN apt-get update -y \
&& apt-get install -y python3-pip && apt-get install -y python3-pip
RUN ldconfig /usr/local/cuda-12.8/compat/ RUN ldconfig /usr/local/cuda-12.9/compat/
# Install vLLM with FlashInfer - use CUDA 12.8 PyTorch wheels (compatible with vLLM 0.15.0) # Install vLLM with FlashInfer - use CUDA 12.8 PyTorch wheels (compatible with vLLM 0.15.0)
RUN python3 -m pip install --upgrade pip && \ RUN python3 -m pip install --upgrade pip && \
python3 -m pip install "vllm[flashinfer]==0.15.0" --extra-index-url https://download.pytorch.org/whl/cu128 python3 -m pip install "vllm[flashinfer]==0.15.0" --extra-index-url https://download.pytorch.org/whl/cu129
@@ -23,6 +23,7 @@ ARG BASE_PATH="/runpod-volume"
ARG QUANTIZATION="" ARG QUANTIZATION=""
ARG MODEL_REVISION="" ARG MODEL_REVISION=""
ARG TOKENIZER_REVISION="" ARG TOKENIZER_REVISION=""
ARG VLLM_NIGHTLY="false"
ENV MODEL_NAME=$MODEL_NAME \ ENV MODEL_NAME=$MODEL_NAME \
MODEL_REVISION=$MODEL_REVISION \ MODEL_REVISION=$MODEL_REVISION \
@@ -44,6 +45,11 @@ ENV MODEL_NAME=$MODEL_NAME \
ENV PYTHONPATH="/:/vllm-workspace" ENV PYTHONPATH="/:/vllm-workspace"
RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \
pip install -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \
apt-get update && apt-get install -y git && rm -rf /var/lib/apt/lists/* && \
pip install git+https://github.com/huggingface/transformers.git; \
fi
COPY src /src COPY src /src
RUN --mount=type=secret,id=HF_TOKEN,required=false \ RUN --mount=type=secret,id=HF_TOKEN,required=false \
+15
View File
@@ -80,6 +80,7 @@ To build an image with the model baked in, you must specify the following docker
- `WORKER_CUDA_VERSION`: `12.1.0` (`12.1.0` is recommended for optimal performance). - `WORKER_CUDA_VERSION`: `12.1.0` (`12.1.0` is recommended for optimal performance).
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer) - `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer)
- `TOKENIZER_REVISION`: Tokenizer revision to load (default: `main`). - `TOKENIZER_REVISION`: Tokenizer revision to load (default: `main`).
- `VLLM_NIGHTLY`: Set to `true` to replace the pinned vLLM release with the latest nightly build and the latest `transformers` from source. Useful for testing unreleased vLLM features. (default: `false`)
For the remaining settings, you may apply them as environment variables when running the container. Supported environment variables are listed in the [Environment Variables](#environment-variables) section. For the remaining settings, you may apply them as environment variables when running the container. Supported environment variables are listed in the [Environment Variables](#environment-variables) section.
@@ -89,6 +90,20 @@ For the remaining settings, you may apply them as environment variables when run
docker build -t username/image:tag --build-arg MODEL_NAME="openchat/openchat_3.5" --build-arg BASE_PATH="/models" . docker build -t username/image:tag --build-arg MODEL_NAME="openchat/openchat_3.5" --build-arg BASE_PATH="/models" .
``` ```
### Example: Building with vLLM Nightly
To use the latest unreleased vLLM build (installs from the nightly wheel index and `transformers` from source):
```bash
docker build -t username/image:tag --build-arg VLLM_NIGHTLY=true .
```
You can combine it with other arguments:
```bash
docker build -t username/image:tag --build-arg VLLM_NIGHTLY=true --build-arg MODEL_NAME="meta-llama/Llama-3.1-8B-Instruct" --build-arg BASE_PATH="/models" .
```
### (Optional) Including Huggingface Token ### (Optional) Including Huggingface Token
If the model you would like to deploy is private or gated, you will need to include it during build time as a Docker secret, which will protect it from being exposed in the image and on DockerHub. If the model you would like to deploy is private or gated, you will need to include it during build time as a Docker secret, which will protect it from being exposed in the image and on DockerHub.
+25 -15
View File
@@ -60,22 +60,32 @@ Complete guide to all environment variables and configuration options for worker
## Speculative Decoding Settings ## Speculative Decoding Settings
| Variable | Default | Type/Choices | Description | Speculative decoding can be configured in two ways:
| ------------------------------------------------ | ------------------- | --------------------------------------------------- | ----------------------------------------------------------------------------------------- |
| `SCHEDULER_DELAY_FACTOR` | 0.0 | `float` | Apply a delay before scheduling next prompt. |
| `ENABLE_CHUNKED_PREFILL` | False | `bool` | Enable chunked prefill requests. |
| `SPECULATIVE_MODEL` | None | `str` | The name of the draft model to be used in speculative decoding. |
| `NUM_SPECULATIVE_TOKENS` | None | `int` | The number of speculative tokens to sample from the draft model. |
| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | None | `int` | Number of tensor parallel replicas for the draft model. |
| `SPECULATIVE_MAX_MODEL_LEN` | None | `int` | The maximum sequence length supported by the draft model. |
| `SPECULATIVE_DISABLE_BY_BATCH_SIZE` | None | `int` | Disable speculative decoding if the number of enqueue requests is larger than this value. |
| `NGRAM_PROMPT_LOOKUP_MAX` | None | `int` | Max size of window for ngram prompt lookup in speculative decoding. |
| `NGRAM_PROMPT_LOOKUP_MIN` | None | `int` | Min size of window for ngram prompt lookup in speculative decoding. |
| `SPEC_DECODING_ACCEPTANCE_METHOD` | 'rejection_sampler' | ['rejection_sampler', 'typical_acceptance_sampler'] | Specify the acceptance method for draft token verification in speculative decoding. |
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD` | None | `float` | Set the lower bound threshold for the posterior probability of a token to be accepted. |
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA` | None | `float` | A scaling factor for the entropy-based threshold for token acceptance. |
## System Performance Settings ### Option 1: JSON Configuration
Set `SPECULATIVE_CONFIG` to a JSON string with your full speculative decoding configuration:
```bash
SPECULATIVE_CONFIG='{"method": "ngram", "num_speculative_tokens": 5, "prompt_lookup_max": 4}'
```
### Option 2: Individual Environment Variables
| Variable | Default | Type/Choices | Description |
| ---------------------------------------- | ------- | ------------------------------------------------------------------ | ----------------------------------------------------------------------------------------- |
| `SPECULATIVE_METHOD` | None | ['draft_model', 'ngram', 'eagle', 'eagle3', 'medusa', 'mlp_speculator'] | Speculative decoding method to use. |
| `SPECULATIVE_MODEL` | None | `str` | The name of the draft model to be used in speculative decoding. |
| `NUM_SPECULATIVE_TOKENS` | None | `int` | The number of speculative tokens to sample from the draft model. |
| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | None | `int` | Number of tensor parallel replicas for the draft model. |
| `SPECULATIVE_MAX_MODEL_LEN` | None | `int` | The maximum sequence length supported by the draft model. |
| `SPECULATIVE_DISABLE_BY_BATCH_SIZE` | None | `int` | Disable speculative decoding if the number of enqueue requests is larger than this value. |
| `NGRAM_PROMPT_LOOKUP_MAX` | None | `int` | Max size of window for ngram prompt lookup in speculative decoding. |
| `NGRAM_PROMPT_LOOKUP_MIN` | None | `int` | Min size of window for ngram prompt lookup in speculative decoding. |
If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When using individual env vars without `SPECULATIVE_METHOD`, the method is auto-detected from the model name or configuration.
## Scheduling & Performance Settings
| Variable | Default | Type/Choices | Description | | Variable | Default | Type/Choices | Description |
| ------------------------------ | ------- | --------------- | ----------------------------------------------------------------------------------------------------------------------------------- | | ------------------------------ | ------- | --------------- | ----------------------------------------------------------------------------------------------------------------------------------- |
+2 -2
View File
@@ -175,7 +175,7 @@ class vLLMEngine:
class OpenAIvLLMEngine(vLLMEngine): class OpenAIvLLMEngine(vLLMEngine):
def __init__(self, vllm_engine): def __init__(self, vllm_engine):
super().__init__(vllm_engine) super().__init__(vllm_engine)
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.engine_args.model self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.engine_args.served_model_name or self.engine_args.model
self.response_role = os.getenv("OPENAI_RESPONSE_ROLE") or "assistant" self.response_role = os.getenv("OPENAI_RESPONSE_ROLE") or "assistant"
self.lora_adapters = self._load_lora_adapters() self.lora_adapters = self._load_lora_adapters()
@@ -233,7 +233,7 @@ class OpenAIvLLMEngine(vLLMEngine):
async def _initialize_engines(self): async def _initialize_engines(self):
self.model_config = self.llm.model_config self.model_config = self.llm.model_config
self.base_model_paths = [ self.base_model_paths = [
BaseModelPath(name=self.engine_args.model, model_path=self.engine_args.model) BaseModelPath(name=self.served_model_name, model_path=self.engine_args.model)
] ]
self.serving_models = OpenAIServingModels( self.serving_models = OpenAIServingModels(
+134 -5
View File
@@ -100,6 +100,115 @@ DEFAULT_ARGS = {
"disable_logprobs_during_spec_decoding": os.getenv('DISABLE_LOGPROBS_DURING_SPEC_DECODING', None), "disable_logprobs_during_spec_decoding": os.getenv('DISABLE_LOGPROBS_DURING_SPEC_DECODING', None),
"otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None), "otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None),
} }
def get_speculative_config():
"""Build speculative decoding configuration from environment variables.
Supports two modes:
1. Full JSON config via SPECULATIVE_CONFIG env var
2. Individual env vars for common settings
"""
# Option 1: Full JSON configuration
spec_config_json = os.getenv('SPECULATIVE_CONFIG')
if spec_config_json:
try:
config = json.loads(spec_config_json)
logging.info(f"Using speculative config from SPECULATIVE_CONFIG: {config}")
return config
except json.JSONDecodeError as e:
logging.error(f"Failed to parse SPECULATIVE_CONFIG JSON: {e}")
return None
# Option 2: Build config from individual environment variables
spec_method = os.getenv('SPECULATIVE_METHOD')
spec_model = os.getenv('SPECULATIVE_MODEL')
_num_spec_tokens = os.getenv('NUM_SPECULATIVE_TOKENS')
_ngram_max = os.getenv('NGRAM_PROMPT_LOOKUP_MAX')
_ngram_min = os.getenv('NGRAM_PROMPT_LOOKUP_MIN')
# Convert numeric vars to int so '0' (hub.json default) is treated as unset
num_spec_tokens = (int(_num_spec_tokens) or None) if _num_spec_tokens else None
ngram_max = (int(_ngram_max) or None) if _ngram_max else None
ngram_min = (int(_ngram_min) or None) if _ngram_min else None
if not any([spec_method, spec_model, ngram_max]):
return None
config = {}
# Determine method
if spec_method:
config['method'] = spec_method
elif ngram_max and not spec_model:
config['method'] = 'ngram'
elif spec_model:
model_lower = spec_model.lower()
if 'eagle3' in model_lower:
config['method'] = 'eagle3'
elif 'eagle' in model_lower:
config['method'] = 'eagle'
elif 'medusa' in model_lower:
config['method'] = 'medusa'
else:
config['method'] = 'draft_model'
if spec_model:
config['model'] = spec_model
if num_spec_tokens:
config['num_speculative_tokens'] = num_spec_tokens
if ngram_max:
config['prompt_lookup_max'] = ngram_max
if ngram_min:
config['prompt_lookup_min'] = ngram_min
draft_tp = os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE')
if draft_tp:
config['draft_tensor_parallel_size'] = int(draft_tp)
spec_max_len = os.getenv('SPECULATIVE_MAX_MODEL_LEN')
if spec_max_len:
config['max_model_len'] = int(spec_max_len)
disable_batch = os.getenv('SPECULATIVE_DISABLE_BY_BATCH_SIZE')
if disable_batch:
config['disable_by_batch_size'] = int(disable_batch)
spec_quant = os.getenv('SPECULATIVE_QUANTIZATION')
if spec_quant:
config['quantization'] = spec_quant
spec_revision = os.getenv('SPECULATIVE_MODEL_REVISION')
if spec_revision:
config['revision'] = spec_revision
spec_eager = os.getenv('SPECULATIVE_ENFORCE_EAGER')
if spec_eager:
config['enforce_eager'] = spec_eager.lower() == 'true'
if config:
logging.info(f"Built speculative config from env vars: {config}")
return config
return None
def _resolve_max_model_len(model, trust_remote_code=False, revision=None):
"""Resolve max_model_len from the model's HuggingFace config."""
try:
from transformers import AutoConfig
config = AutoConfig.from_pretrained(
model,
trust_remote_code=trust_remote_code,
revision=revision,
)
for attr in ('max_position_embeddings', 'n_positions', 'max_seq_len', 'seq_length'):
val = getattr(config, attr, None)
if val is not None:
logging.info(f"Resolved max_model_len={val} from model config ({attr})")
return val
except Exception as e:
logging.warning(f"Could not resolve max_model_len from model config: {e}")
return None
limit_mm_env = os.getenv('LIMIT_MM_PER_PROMPT') limit_mm_env = os.getenv('LIMIT_MM_PER_PROMPT')
if limit_mm_env is not None: if limit_mm_env is not None:
DEFAULT_ARGS["limit_mm_per_prompt"] = convert_limit_mm_per_prompt(limit_mm_env) DEFAULT_ARGS["limit_mm_per_prompt"] = convert_limit_mm_per_prompt(limit_mm_env)
@@ -182,11 +291,26 @@ def get_engine_args():
# os.environ["VLLM_ATTENTION_BACKEND"] = "FLASHINFER" # os.environ["VLLM_ATTENTION_BACKEND"] = "FLASHINFER"
# logging.info("Using FLASHINFER for gemma-2 model.") # logging.info("Using FLASHINFER for gemma-2 model.")
# When max_num_batched_tokens is None (env var was 0), set to max_model_len # Set max_num_batched_tokens to max_model_len for unlimited batching.
# to preserve "unlimited" behavior. vLLM defaults None to 2048. # vLLM defaults max_num_batched_tokens to 2048 when None, which is too low.
if args.get("max_num_batched_tokens") is None and args.get("max_model_len") is not None:
args["max_num_batched_tokens"] = args["max_model_len"] if args.get("max_model_len") == 0:
logging.info(f"Setting max_num_batched_tokens to max_model_len ({args['max_model_len']}) for unlimited batching.") args["max_model_len"] = None
if args.get("max_num_batched_tokens") == 0:
args["max_num_batched_tokens"] = None
if args.get("max_num_batched_tokens") is None:
max_model_len = args.get("max_model_len")
if max_model_len is None:
max_model_len = _resolve_max_model_len(
args.get("model"),
trust_remote_code=args.get("trust_remote_code", False),
revision=args.get("revision"),
)
if max_model_len is not None:
args["max_num_batched_tokens"] = max_model_len
logging.info(f"Setting max_num_batched_tokens to {max_model_len}")
# VLLM_ATTENTION_BACKEND is deprecated, migrate to attention_backend # VLLM_ATTENTION_BACKEND is deprecated, migrate to attention_backend
if os.getenv('VLLM_ATTENTION_BACKEND'): if os.getenv('VLLM_ATTENTION_BACKEND'):
@@ -207,4 +331,9 @@ def get_engine_args():
if os.getenv('DISABLE_LOG_REQUESTS', 'False').lower() == 'true': if os.getenv('DISABLE_LOG_REQUESTS', 'False').lower() == 'true':
args['enable_log_requests'] = False args['enable_log_requests'] = False
# Add speculative decoding configuration if present
speculative_config = get_speculative_config()
if speculative_config:
args["speculative_config"] = speculative_config
return AsyncEngineArgs(**args) return AsyncEngineArgs(**args)
+4 -4
View File
@@ -6,7 +6,7 @@ from time import time
try: try:
from vllm.utils import random_uuid from vllm.utils import random_uuid
from vllm.entrypoints.openai.engine.protocol import ErrorResponse, RequestResponseMetadata from vllm.entrypoints.openai.engine.protocol import ErrorResponse, ErrorInfo, RequestResponseMetadata
from vllm import SamplingParams from vllm import SamplingParams
except ImportError: except ImportError:
logging.warning("Error importing vllm, skipping related imports. This is ONLY expected when baking model into docker image from a machine without GPUs") logging.warning("Error importing vllm, skipping related imports. This is ONLY expected when baking model into docker image from a machine without GPUs")
@@ -87,9 +87,9 @@ class BatchSize:
self.current_batch_size = min(self.current_batch_size*self.batch_size_growth_factor, self.max_batch_size) self.current_batch_size = min(self.current_batch_size*self.batch_size_growth_factor, self.max_batch_size)
def create_error_response(message: str, err_type: str = "BadRequestError", status_code: HTTPStatus = HTTPStatus.BAD_REQUEST) -> ErrorResponse: def create_error_response(message: str, err_type: str = "BadRequestError", status_code: HTTPStatus = HTTPStatus.BAD_REQUEST) -> ErrorResponse:
return ErrorResponse(message=message, return ErrorResponse(error=ErrorInfo(message=message,
type=err_type, type=err_type,
code=status_code.value) code=status_code.value))
def get_int_bool_env(env_var: str, default: bool) -> bool: def get_int_bool_env(env_var: str, default: bool) -> bool:
return int(os.getenv(env_var, int(default))) == 1 return int(os.getenv(env_var, int(default))) == 1