Compare commits

..
14 Commits
Author SHA1 Message Date
chrisvelaandGitHub 1606cff557 Merge pull request #265 from runpod-workers/fix/zero-max-model-num_batches
Release / release (push) Waiting to run
fix: check for zero param and set to None
2026-02-13 15:26:06 -06:00
velaraptor-runpod e705c9494b fix: check for zero param and set to None 2026-02-13 15:23:54 -06:00
chrisvelaandGitHub b749aa5718 Merge pull request #264 from runpod-workers/fix/max_num_batched_tokens
Release / release (push) Waiting to run
fix: max num batched tokens
2026-02-13 12:38:01 -06:00
velaraptor-runpod 4705ba8a7c fix: check max_num_batched_tokenz if max_model_len not set 2026-02-13 03:29:52 -06:00
velaraptor-runpod 767c66c301 make minimal changes 2026-02-13 03:23:44 -06:00
velaraptor-runpod fefdbe21a9 update changes 2026-02-13 03:16:43 -06:00
velaraptor-runpod ee961ad28d Update hub.json 2026-02-13 03:08:19 -06:00
velaraptor-runpod 2e8c251447 Merge branch 'main' into feat/update-vllm-v0.15.0 2026-02-13 03:01:05 -06:00
velaraptor-runpod c3cf43b228 Update hub.json 2026-02-13 00:22:16 -06:00
velaraptor-runpod 7ec10b98cd Update utils.py 2026-02-12 15:28:31 -06:00
velaraptor-runpod 340bc0b3c6 fix: served model name 2026-02-10 21:42:58 -06:00
velaraptor-runpod e1e9ef74ad add changes from pr 2026-02-06 18:10:09 -06:00
velaraptor-runpod 461f89cea6 add torch-c-dlpack-ext requirement 2026-02-06 17:03:39 -06:00
velaraptor-runpod 8eb55b90c1 add changes for v0.15.0 2026-02-05 17:24:16 -06:00
6 changed files with 225 additions and 27 deletions
+59 -1
View File
@@ -187,6 +187,7 @@
"name": "Max Model Length",
"type": "number",
"description": "Model context length.",
"default": null,
"advanced": true
}
},
@@ -206,7 +207,8 @@
"value": "mp"
}
],
"advanced": true
"advanced": true,
"default": "mp"
}
},
{
@@ -293,6 +295,7 @@
"name": "Max Num Batched Tokens",
"type": "number",
"description": "Maximum number of batched tokens per iteration.",
"default": null,
"advanced": true
}
},
@@ -490,6 +493,61 @@
"advanced": true
}
},
{
"key": "SPECULATIVE_CONFIG",
"input": {
"name": "Speculative Config (JSON)",
"type": "string",
"description": "Full speculative decoding configuration as a JSON string. Overrides individual speculative env vars.",
"advanced": true
}
},
{
"key": "SPECULATIVE_METHOD",
"input": {
"name": "Speculative Method",
"type": "string",
"description": "Speculative decoding method to use.",
"options": [
{ "label": "None", "value": "" },
{ "label": "Draft Model", "value": "draft_model" },
{ "label": "N-gram", "value": "ngram" },
{ "label": "EAGLE", "value": "eagle" },
{ "label": "EAGLE3", "value": "eagle3" },
{ "label": "Medusa", "value": "medusa" },
{ "label": "MLP Speculator", "value": "mlp_speculator" }
],
"default": "",
"advanced": true
}
},
{
"key": "SPECULATIVE_MODEL",
"input": {
"name": "Speculative Model",
"type": "string",
"description": "The name of the draft model to be used in speculative decoding.",
"advanced": true
}
},
{
"key": "NUM_SPECULATIVE_TOKENS",
"input": {
"name": "Num Speculative Tokens",
"type": "number",
"description": "The number of speculative tokens to sample from the draft model.",
"advanced": true
}
},
{
"key": "NGRAM_PROMPT_LOOKUP_MAX",
"input": {
"name": "Ngram Prompt Lookup Max",
"type": "number",
"description": "Max size of window for ngram prompt lookup in speculative decoding.",
"advanced": true
}
},
{
"key": "MODEL_LOADER_EXTRA_CONFIG",
"input": {
+6
View File
@@ -23,6 +23,7 @@ ARG BASE_PATH="/runpod-volume"
ARG QUANTIZATION=""
ARG MODEL_REVISION=""
ARG TOKENIZER_REVISION=""
ARG VLLM_NIGHTLY="false"
ENV MODEL_NAME=$MODEL_NAME \
MODEL_REVISION=$MODEL_REVISION \
@@ -44,6 +45,11 @@ ENV MODEL_NAME=$MODEL_NAME \
ENV PYTHONPATH="/:/vllm-workspace"
RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \
pip install -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \
apt-get update && apt-get install -y git && rm -rf /var/lib/apt/lists/* && \
pip install git+https://github.com/huggingface/transformers.git; \
fi
COPY src /src
RUN --mount=type=secret,id=HF_TOKEN,required=false \
+25 -15
View File
@@ -60,22 +60,32 @@ Complete guide to all environment variables and configuration options for worker
## Speculative Decoding Settings
| Variable | Default | Type/Choices | Description |
| ------------------------------------------------ | ------------------- | --------------------------------------------------- | ----------------------------------------------------------------------------------------- |
| `SCHEDULER_DELAY_FACTOR` | 0.0 | `float` | Apply a delay before scheduling next prompt. |
| `ENABLE_CHUNKED_PREFILL` | False | `bool` | Enable chunked prefill requests. |
| `SPECULATIVE_MODEL` | None | `str` | The name of the draft model to be used in speculative decoding. |
| `NUM_SPECULATIVE_TOKENS` | None | `int` | The number of speculative tokens to sample from the draft model. |
| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | None | `int` | Number of tensor parallel replicas for the draft model. |
| `SPECULATIVE_MAX_MODEL_LEN` | None | `int` | The maximum sequence length supported by the draft model. |
| `SPECULATIVE_DISABLE_BY_BATCH_SIZE` | None | `int` | Disable speculative decoding if the number of enqueue requests is larger than this value. |
| `NGRAM_PROMPT_LOOKUP_MAX` | None | `int` | Max size of window for ngram prompt lookup in speculative decoding. |
| `NGRAM_PROMPT_LOOKUP_MIN` | None | `int` | Min size of window for ngram prompt lookup in speculative decoding. |
| `SPEC_DECODING_ACCEPTANCE_METHOD` | 'rejection_sampler' | ['rejection_sampler', 'typical_acceptance_sampler'] | Specify the acceptance method for draft token verification in speculative decoding. |
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD` | None | `float` | Set the lower bound threshold for the posterior probability of a token to be accepted. |
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA` | None | `float` | A scaling factor for the entropy-based threshold for token acceptance. |
Speculative decoding can be configured in two ways:
## System Performance Settings
### Option 1: JSON Configuration
Set `SPECULATIVE_CONFIG` to a JSON string with your full speculative decoding configuration:
```bash
SPECULATIVE_CONFIG='{"method": "ngram", "num_speculative_tokens": 5, "prompt_lookup_max": 4}'
```
### Option 2: Individual Environment Variables
| Variable | Default | Type/Choices | Description |
| ---------------------------------------- | ------- | ------------------------------------------------------------------ | ----------------------------------------------------------------------------------------- |
| `SPECULATIVE_METHOD` | None | ['draft_model', 'ngram', 'eagle', 'eagle3', 'medusa', 'mlp_speculator'] | Speculative decoding method to use. |
| `SPECULATIVE_MODEL` | None | `str` | The name of the draft model to be used in speculative decoding. |
| `NUM_SPECULATIVE_TOKENS` | None | `int` | The number of speculative tokens to sample from the draft model. |
| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | None | `int` | Number of tensor parallel replicas for the draft model. |
| `SPECULATIVE_MAX_MODEL_LEN` | None | `int` | The maximum sequence length supported by the draft model. |
| `SPECULATIVE_DISABLE_BY_BATCH_SIZE` | None | `int` | Disable speculative decoding if the number of enqueue requests is larger than this value. |
| `NGRAM_PROMPT_LOOKUP_MAX` | None | `int` | Max size of window for ngram prompt lookup in speculative decoding. |
| `NGRAM_PROMPT_LOOKUP_MIN` | None | `int` | Min size of window for ngram prompt lookup in speculative decoding. |
If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When using individual env vars without `SPECULATIVE_METHOD`, the method is auto-detected from the model name or configuration.
## Scheduling & Performance Settings
| Variable | Default | Type/Choices | Description |
| ------------------------------ | ------- | --------------- | ----------------------------------------------------------------------------------------------------------------------------------- |
+2 -2
View File
@@ -175,7 +175,7 @@ class vLLMEngine:
class OpenAIvLLMEngine(vLLMEngine):
def __init__(self, vllm_engine):
super().__init__(vllm_engine)
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.engine_args.model
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.engine_args.served_model_name or self.engine_args.model
self.response_role = os.getenv("OPENAI_RESPONSE_ROLE") or "assistant"
self.lora_adapters = self._load_lora_adapters()
@@ -233,7 +233,7 @@ class OpenAIvLLMEngine(vLLMEngine):
async def _initialize_engines(self):
self.model_config = self.llm.model_config
self.base_model_paths = [
BaseModelPath(name=self.engine_args.model, model_path=self.engine_args.model)
BaseModelPath(name=self.served_model_name, model_path=self.engine_args.model)
]
self.serving_models = OpenAIServingModels(
+129 -5
View File
@@ -100,6 +100,110 @@ DEFAULT_ARGS = {
"disable_logprobs_during_spec_decoding": os.getenv('DISABLE_LOGPROBS_DURING_SPEC_DECODING', None),
"otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None),
}
def get_speculative_config():
"""Build speculative decoding configuration from environment variables.
Supports two modes:
1. Full JSON config via SPECULATIVE_CONFIG env var
2. Individual env vars for common settings
"""
# Option 1: Full JSON configuration
spec_config_json = os.getenv('SPECULATIVE_CONFIG')
if spec_config_json:
try:
config = json.loads(spec_config_json)
logging.info(f"Using speculative config from SPECULATIVE_CONFIG: {config}")
return config
except json.JSONDecodeError as e:
logging.error(f"Failed to parse SPECULATIVE_CONFIG JSON: {e}")
return None
# Option 2: Build config from individual environment variables
spec_method = os.getenv('SPECULATIVE_METHOD')
spec_model = os.getenv('SPECULATIVE_MODEL')
num_spec_tokens = os.getenv('NUM_SPECULATIVE_TOKENS')
ngram_max = os.getenv('NGRAM_PROMPT_LOOKUP_MAX')
ngram_min = os.getenv('NGRAM_PROMPT_LOOKUP_MIN')
if not any([spec_method, spec_model, ngram_max]):
return None
config = {}
# Determine method
if spec_method:
config['method'] = spec_method
elif ngram_max and not spec_model:
config['method'] = 'ngram'
elif spec_model:
model_lower = spec_model.lower()
if 'eagle3' in model_lower:
config['method'] = 'eagle3'
elif 'eagle' in model_lower:
config['method'] = 'eagle'
elif 'medusa' in model_lower:
config['method'] = 'medusa'
else:
config['method'] = 'draft_model'
if spec_model:
config['model'] = spec_model
if num_spec_tokens:
config['num_speculative_tokens'] = int(num_spec_tokens)
if ngram_max:
config['prompt_lookup_max'] = int(ngram_max)
if ngram_min:
config['prompt_lookup_min'] = int(ngram_min)
draft_tp = os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE')
if draft_tp:
config['draft_tensor_parallel_size'] = int(draft_tp)
spec_max_len = os.getenv('SPECULATIVE_MAX_MODEL_LEN')
if spec_max_len:
config['max_model_len'] = int(spec_max_len)
disable_batch = os.getenv('SPECULATIVE_DISABLE_BY_BATCH_SIZE')
if disable_batch:
config['disable_by_batch_size'] = int(disable_batch)
spec_quant = os.getenv('SPECULATIVE_QUANTIZATION')
if spec_quant:
config['quantization'] = spec_quant
spec_revision = os.getenv('SPECULATIVE_MODEL_REVISION')
if spec_revision:
config['revision'] = spec_revision
spec_eager = os.getenv('SPECULATIVE_ENFORCE_EAGER')
if spec_eager:
config['enforce_eager'] = spec_eager.lower() == 'true'
if config:
logging.info(f"Built speculative config from env vars: {config}")
return config
return None
def _resolve_max_model_len(model, trust_remote_code=False, revision=None):
"""Resolve max_model_len from the model's HuggingFace config."""
try:
from transformers import AutoConfig
config = AutoConfig.from_pretrained(
model,
trust_remote_code=trust_remote_code,
revision=revision,
)
for attr in ('max_position_embeddings', 'n_positions', 'max_seq_len', 'seq_length'):
val = getattr(config, attr, None)
if val is not None:
logging.info(f"Resolved max_model_len={val} from model config ({attr})")
return val
except Exception as e:
logging.warning(f"Could not resolve max_model_len from model config: {e}")
return None
limit_mm_env = os.getenv('LIMIT_MM_PER_PROMPT')
if limit_mm_env is not None:
DEFAULT_ARGS["limit_mm_per_prompt"] = convert_limit_mm_per_prompt(limit_mm_env)
@@ -182,11 +286,26 @@ def get_engine_args():
# os.environ["VLLM_ATTENTION_BACKEND"] = "FLASHINFER"
# logging.info("Using FLASHINFER for gemma-2 model.")
# When max_num_batched_tokens is None (env var was 0), set to max_model_len
# to preserve "unlimited" behavior. vLLM defaults None to 2048.
if args.get("max_num_batched_tokens") is None and args.get("max_model_len") is not None:
args["max_num_batched_tokens"] = args["max_model_len"]
logging.info(f"Setting max_num_batched_tokens to max_model_len ({args['max_model_len']}) for unlimited batching.")
# Set max_num_batched_tokens to max_model_len for unlimited batching.
# vLLM defaults max_num_batched_tokens to 2048 when None, which is too low.
if args.get("max_model_len") == 0:
args["max_model_len"] = None
if args.get("max_num_batched_tokens") == 0:
args["max_num_batched_tokens"] = None
if args.get("max_num_batched_tokens") is None:
max_model_len = args.get("max_model_len")
if max_model_len is None:
max_model_len = _resolve_max_model_len(
args.get("model"),
trust_remote_code=args.get("trust_remote_code", False),
revision=args.get("revision"),
)
if max_model_len is not None:
args["max_num_batched_tokens"] = max_model_len
logging.info(f"Setting max_num_batched_tokens to {max_model_len}")
# VLLM_ATTENTION_BACKEND is deprecated, migrate to attention_backend
if os.getenv('VLLM_ATTENTION_BACKEND'):
@@ -207,4 +326,9 @@ def get_engine_args():
if os.getenv('DISABLE_LOG_REQUESTS', 'False').lower() == 'true':
args['enable_log_requests'] = False
# Add speculative decoding configuration if present
speculative_config = get_speculative_config()
if speculative_config:
args["speculative_config"] = speculative_config
return AsyncEngineArgs(**args)
+4 -4
View File
@@ -6,7 +6,7 @@ from time import time
try:
from vllm.utils import random_uuid
from vllm.entrypoints.openai.engine.protocol import ErrorResponse, RequestResponseMetadata
from vllm.entrypoints.openai.engine.protocol import ErrorResponse, ErrorInfo, RequestResponseMetadata
from vllm import SamplingParams
except ImportError:
logging.warning("Error importing vllm, skipping related imports. This is ONLY expected when baking model into docker image from a machine without GPUs")
@@ -87,9 +87,9 @@ class BatchSize:
self.current_batch_size = min(self.current_batch_size*self.batch_size_growth_factor, self.max_batch_size)
def create_error_response(message: str, err_type: str = "BadRequestError", status_code: HTTPStatus = HTTPStatus.BAD_REQUEST) -> ErrorResponse:
return ErrorResponse(message=message,
type=err_type,
code=status_code.value)
return ErrorResponse(error=ErrorInfo(message=message,
type=err_type,
code=status_code.value))
def get_int_bool_env(env_var: str, default: bool) -> bool:
return int(os.getenv(env_var, int(default))) == 1