Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d69cc021e8 | ||
|
|
61faa8f137 | ||
|
|
1606cff557 | ||
|
|
e705c9494b | ||
|
|
b749aa5718 | ||
|
|
4705ba8a7c | ||
|
|
767c66c301 | ||
|
|
fefdbe21a9 | ||
|
|
ee961ad28d | ||
|
|
2e8c251447 | ||
|
|
c3cf43b228 | ||
|
|
7ec10b98cd | ||
|
|
340bc0b3c6 | ||
|
|
e1e9ef74ad | ||
|
|
461f89cea6 | ||
|
|
8eb55b90c1 |
+59
-1
@@ -187,6 +187,7 @@
|
||||
"name": "Max Model Length",
|
||||
"type": "number",
|
||||
"description": "Model context length.",
|
||||
"default": null,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
@@ -206,7 +207,8 @@
|
||||
"value": "mp"
|
||||
}
|
||||
],
|
||||
"advanced": true
|
||||
"advanced": true,
|
||||
"default": "mp"
|
||||
}
|
||||
},
|
||||
{
|
||||
@@ -293,6 +295,7 @@
|
||||
"name": "Max Num Batched Tokens",
|
||||
"type": "number",
|
||||
"description": "Maximum number of batched tokens per iteration.",
|
||||
"default": null,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
@@ -490,6 +493,61 @@
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "SPECULATIVE_CONFIG",
|
||||
"input": {
|
||||
"name": "Speculative Config (JSON)",
|
||||
"type": "string",
|
||||
"description": "Full speculative decoding configuration as a JSON string. Overrides individual speculative env vars.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "SPECULATIVE_METHOD",
|
||||
"input": {
|
||||
"name": "Speculative Method",
|
||||
"type": "string",
|
||||
"description": "Speculative decoding method to use.",
|
||||
"options": [
|
||||
{ "label": "None", "value": "" },
|
||||
{ "label": "Draft Model", "value": "draft_model" },
|
||||
{ "label": "N-gram", "value": "ngram" },
|
||||
{ "label": "EAGLE", "value": "eagle" },
|
||||
{ "label": "EAGLE3", "value": "eagle3" },
|
||||
{ "label": "Medusa", "value": "medusa" },
|
||||
{ "label": "MLP Speculator", "value": "mlp_speculator" }
|
||||
],
|
||||
"default": "",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "SPECULATIVE_MODEL",
|
||||
"input": {
|
||||
"name": "Speculative Model",
|
||||
"type": "string",
|
||||
"description": "The name of the draft model to be used in speculative decoding.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "NUM_SPECULATIVE_TOKENS",
|
||||
"input": {
|
||||
"name": "Num Speculative Tokens",
|
||||
"type": "number",
|
||||
"description": "The number of speculative tokens to sample from the draft model.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "NGRAM_PROMPT_LOOKUP_MAX",
|
||||
"input": {
|
||||
"name": "Ngram Prompt Lookup Max",
|
||||
"type": "number",
|
||||
"description": "Max size of window for ngram prompt lookup in speculative decoding.",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "MODEL_LOADER_EXTRA_CONFIG",
|
||||
"input": {
|
||||
|
||||
@@ -23,6 +23,7 @@ ARG BASE_PATH="/runpod-volume"
|
||||
ARG QUANTIZATION=""
|
||||
ARG MODEL_REVISION=""
|
||||
ARG TOKENIZER_REVISION=""
|
||||
ARG VLLM_NIGHTLY="false"
|
||||
|
||||
ENV MODEL_NAME=$MODEL_NAME \
|
||||
MODEL_REVISION=$MODEL_REVISION \
|
||||
@@ -44,6 +45,11 @@ ENV MODEL_NAME=$MODEL_NAME \
|
||||
|
||||
ENV PYTHONPATH="/:/vllm-workspace"
|
||||
|
||||
RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \
|
||||
pip install -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \
|
||||
apt-get update && apt-get install -y git && rm -rf /var/lib/apt/lists/* && \
|
||||
pip install git+https://github.com/huggingface/transformers.git; \
|
||||
fi
|
||||
|
||||
COPY src /src
|
||||
RUN --mount=type=secret,id=HF_TOKEN,required=false \
|
||||
|
||||
+25
-15
@@ -60,22 +60,32 @@ Complete guide to all environment variables and configuration options for worker
|
||||
|
||||
## Speculative Decoding Settings
|
||||
|
||||
| Variable | Default | Type/Choices | Description |
|
||||
| ------------------------------------------------ | ------------------- | --------------------------------------------------- | ----------------------------------------------------------------------------------------- |
|
||||
| `SCHEDULER_DELAY_FACTOR` | 0.0 | `float` | Apply a delay before scheduling next prompt. |
|
||||
| `ENABLE_CHUNKED_PREFILL` | False | `bool` | Enable chunked prefill requests. |
|
||||
| `SPECULATIVE_MODEL` | None | `str` | The name of the draft model to be used in speculative decoding. |
|
||||
| `NUM_SPECULATIVE_TOKENS` | None | `int` | The number of speculative tokens to sample from the draft model. |
|
||||
| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | None | `int` | Number of tensor parallel replicas for the draft model. |
|
||||
| `SPECULATIVE_MAX_MODEL_LEN` | None | `int` | The maximum sequence length supported by the draft model. |
|
||||
| `SPECULATIVE_DISABLE_BY_BATCH_SIZE` | None | `int` | Disable speculative decoding if the number of enqueue requests is larger than this value. |
|
||||
| `NGRAM_PROMPT_LOOKUP_MAX` | None | `int` | Max size of window for ngram prompt lookup in speculative decoding. |
|
||||
| `NGRAM_PROMPT_LOOKUP_MIN` | None | `int` | Min size of window for ngram prompt lookup in speculative decoding. |
|
||||
| `SPEC_DECODING_ACCEPTANCE_METHOD` | 'rejection_sampler' | ['rejection_sampler', 'typical_acceptance_sampler'] | Specify the acceptance method for draft token verification in speculative decoding. |
|
||||
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD` | None | `float` | Set the lower bound threshold for the posterior probability of a token to be accepted. |
|
||||
| `TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA` | None | `float` | A scaling factor for the entropy-based threshold for token acceptance. |
|
||||
Speculative decoding can be configured in two ways:
|
||||
|
||||
## System Performance Settings
|
||||
### Option 1: JSON Configuration
|
||||
|
||||
Set `SPECULATIVE_CONFIG` to a JSON string with your full speculative decoding configuration:
|
||||
|
||||
```bash
|
||||
SPECULATIVE_CONFIG='{"method": "ngram", "num_speculative_tokens": 5, "prompt_lookup_max": 4}'
|
||||
```
|
||||
|
||||
### Option 2: Individual Environment Variables
|
||||
|
||||
| Variable | Default | Type/Choices | Description |
|
||||
| ---------------------------------------- | ------- | ------------------------------------------------------------------ | ----------------------------------------------------------------------------------------- |
|
||||
| `SPECULATIVE_METHOD` | None | ['draft_model', 'ngram', 'eagle', 'eagle3', 'medusa', 'mlp_speculator'] | Speculative decoding method to use. |
|
||||
| `SPECULATIVE_MODEL` | None | `str` | The name of the draft model to be used in speculative decoding. |
|
||||
| `NUM_SPECULATIVE_TOKENS` | None | `int` | The number of speculative tokens to sample from the draft model. |
|
||||
| `SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE` | None | `int` | Number of tensor parallel replicas for the draft model. |
|
||||
| `SPECULATIVE_MAX_MODEL_LEN` | None | `int` | The maximum sequence length supported by the draft model. |
|
||||
| `SPECULATIVE_DISABLE_BY_BATCH_SIZE` | None | `int` | Disable speculative decoding if the number of enqueue requests is larger than this value. |
|
||||
| `NGRAM_PROMPT_LOOKUP_MAX` | None | `int` | Max size of window for ngram prompt lookup in speculative decoding. |
|
||||
| `NGRAM_PROMPT_LOOKUP_MIN` | None | `int` | Min size of window for ngram prompt lookup in speculative decoding. |
|
||||
|
||||
If `SPECULATIVE_CONFIG` is set, it takes priority over individual env vars. When using individual env vars without `SPECULATIVE_METHOD`, the method is auto-detected from the model name or configuration.
|
||||
|
||||
## Scheduling & Performance Settings
|
||||
|
||||
| Variable | Default | Type/Choices | Description |
|
||||
| ------------------------------ | ------- | --------------- | ----------------------------------------------------------------------------------------------------------------------------------- |
|
||||
|
||||
+2
-2
@@ -175,7 +175,7 @@ class vLLMEngine:
|
||||
class OpenAIvLLMEngine(vLLMEngine):
|
||||
def __init__(self, vllm_engine):
|
||||
super().__init__(vllm_engine)
|
||||
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.engine_args.model
|
||||
self.served_model_name = os.getenv("OPENAI_SERVED_MODEL_NAME_OVERRIDE") or self.engine_args.served_model_name or self.engine_args.model
|
||||
self.response_role = os.getenv("OPENAI_RESPONSE_ROLE") or "assistant"
|
||||
self.lora_adapters = self._load_lora_adapters()
|
||||
|
||||
@@ -233,7 +233,7 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
async def _initialize_engines(self):
|
||||
self.model_config = self.llm.model_config
|
||||
self.base_model_paths = [
|
||||
BaseModelPath(name=self.engine_args.model, model_path=self.engine_args.model)
|
||||
BaseModelPath(name=self.served_model_name, model_path=self.engine_args.model)
|
||||
]
|
||||
|
||||
self.serving_models = OpenAIServingModels(
|
||||
|
||||
+134
-5
@@ -100,6 +100,115 @@ DEFAULT_ARGS = {
|
||||
"disable_logprobs_during_spec_decoding": os.getenv('DISABLE_LOGPROBS_DURING_SPEC_DECODING', None),
|
||||
"otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None),
|
||||
}
|
||||
|
||||
def get_speculative_config():
|
||||
"""Build speculative decoding configuration from environment variables.
|
||||
|
||||
Supports two modes:
|
||||
1. Full JSON config via SPECULATIVE_CONFIG env var
|
||||
2. Individual env vars for common settings
|
||||
"""
|
||||
# Option 1: Full JSON configuration
|
||||
spec_config_json = os.getenv('SPECULATIVE_CONFIG')
|
||||
if spec_config_json:
|
||||
try:
|
||||
config = json.loads(spec_config_json)
|
||||
logging.info(f"Using speculative config from SPECULATIVE_CONFIG: {config}")
|
||||
return config
|
||||
except json.JSONDecodeError as e:
|
||||
logging.error(f"Failed to parse SPECULATIVE_CONFIG JSON: {e}")
|
||||
return None
|
||||
|
||||
# Option 2: Build config from individual environment variables
|
||||
spec_method = os.getenv('SPECULATIVE_METHOD')
|
||||
spec_model = os.getenv('SPECULATIVE_MODEL')
|
||||
_num_spec_tokens = os.getenv('NUM_SPECULATIVE_TOKENS')
|
||||
_ngram_max = os.getenv('NGRAM_PROMPT_LOOKUP_MAX')
|
||||
_ngram_min = os.getenv('NGRAM_PROMPT_LOOKUP_MIN')
|
||||
|
||||
# Convert numeric vars to int so '0' (hub.json default) is treated as unset
|
||||
num_spec_tokens = (int(_num_spec_tokens) or None) if _num_spec_tokens else None
|
||||
ngram_max = (int(_ngram_max) or None) if _ngram_max else None
|
||||
ngram_min = (int(_ngram_min) or None) if _ngram_min else None
|
||||
|
||||
if not any([spec_method, spec_model, ngram_max]):
|
||||
return None
|
||||
|
||||
config = {}
|
||||
|
||||
# Determine method
|
||||
if spec_method:
|
||||
config['method'] = spec_method
|
||||
elif ngram_max and not spec_model:
|
||||
config['method'] = 'ngram'
|
||||
elif spec_model:
|
||||
model_lower = spec_model.lower()
|
||||
if 'eagle3' in model_lower:
|
||||
config['method'] = 'eagle3'
|
||||
elif 'eagle' in model_lower:
|
||||
config['method'] = 'eagle'
|
||||
elif 'medusa' in model_lower:
|
||||
config['method'] = 'medusa'
|
||||
else:
|
||||
config['method'] = 'draft_model'
|
||||
|
||||
if spec_model:
|
||||
config['model'] = spec_model
|
||||
if num_spec_tokens:
|
||||
config['num_speculative_tokens'] = num_spec_tokens
|
||||
if ngram_max:
|
||||
config['prompt_lookup_max'] = ngram_max
|
||||
if ngram_min:
|
||||
config['prompt_lookup_min'] = ngram_min
|
||||
|
||||
draft_tp = os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE')
|
||||
if draft_tp:
|
||||
config['draft_tensor_parallel_size'] = int(draft_tp)
|
||||
|
||||
spec_max_len = os.getenv('SPECULATIVE_MAX_MODEL_LEN')
|
||||
if spec_max_len:
|
||||
config['max_model_len'] = int(spec_max_len)
|
||||
|
||||
disable_batch = os.getenv('SPECULATIVE_DISABLE_BY_BATCH_SIZE')
|
||||
if disable_batch:
|
||||
config['disable_by_batch_size'] = int(disable_batch)
|
||||
|
||||
spec_quant = os.getenv('SPECULATIVE_QUANTIZATION')
|
||||
if spec_quant:
|
||||
config['quantization'] = spec_quant
|
||||
|
||||
spec_revision = os.getenv('SPECULATIVE_MODEL_REVISION')
|
||||
if spec_revision:
|
||||
config['revision'] = spec_revision
|
||||
|
||||
spec_eager = os.getenv('SPECULATIVE_ENFORCE_EAGER')
|
||||
if spec_eager:
|
||||
config['enforce_eager'] = spec_eager.lower() == 'true'
|
||||
|
||||
if config:
|
||||
logging.info(f"Built speculative config from env vars: {config}")
|
||||
return config
|
||||
|
||||
return None
|
||||
|
||||
def _resolve_max_model_len(model, trust_remote_code=False, revision=None):
|
||||
"""Resolve max_model_len from the model's HuggingFace config."""
|
||||
try:
|
||||
from transformers import AutoConfig
|
||||
config = AutoConfig.from_pretrained(
|
||||
model,
|
||||
trust_remote_code=trust_remote_code,
|
||||
revision=revision,
|
||||
)
|
||||
for attr in ('max_position_embeddings', 'n_positions', 'max_seq_len', 'seq_length'):
|
||||
val = getattr(config, attr, None)
|
||||
if val is not None:
|
||||
logging.info(f"Resolved max_model_len={val} from model config ({attr})")
|
||||
return val
|
||||
except Exception as e:
|
||||
logging.warning(f"Could not resolve max_model_len from model config: {e}")
|
||||
return None
|
||||
|
||||
limit_mm_env = os.getenv('LIMIT_MM_PER_PROMPT')
|
||||
if limit_mm_env is not None:
|
||||
DEFAULT_ARGS["limit_mm_per_prompt"] = convert_limit_mm_per_prompt(limit_mm_env)
|
||||
@@ -182,11 +291,26 @@ def get_engine_args():
|
||||
# os.environ["VLLM_ATTENTION_BACKEND"] = "FLASHINFER"
|
||||
# logging.info("Using FLASHINFER for gemma-2 model.")
|
||||
|
||||
# When max_num_batched_tokens is None (env var was 0), set to max_model_len
|
||||
# to preserve "unlimited" behavior. vLLM defaults None to 2048.
|
||||
if args.get("max_num_batched_tokens") is None and args.get("max_model_len") is not None:
|
||||
args["max_num_batched_tokens"] = args["max_model_len"]
|
||||
logging.info(f"Setting max_num_batched_tokens to max_model_len ({args['max_model_len']}) for unlimited batching.")
|
||||
# Set max_num_batched_tokens to max_model_len for unlimited batching.
|
||||
# vLLM defaults max_num_batched_tokens to 2048 when None, which is too low.
|
||||
|
||||
if args.get("max_model_len") == 0:
|
||||
args["max_model_len"] = None
|
||||
|
||||
if args.get("max_num_batched_tokens") == 0:
|
||||
args["max_num_batched_tokens"] = None
|
||||
|
||||
if args.get("max_num_batched_tokens") is None:
|
||||
max_model_len = args.get("max_model_len")
|
||||
if max_model_len is None:
|
||||
max_model_len = _resolve_max_model_len(
|
||||
args.get("model"),
|
||||
trust_remote_code=args.get("trust_remote_code", False),
|
||||
revision=args.get("revision"),
|
||||
)
|
||||
if max_model_len is not None:
|
||||
args["max_num_batched_tokens"] = max_model_len
|
||||
logging.info(f"Setting max_num_batched_tokens to {max_model_len}")
|
||||
|
||||
# VLLM_ATTENTION_BACKEND is deprecated, migrate to attention_backend
|
||||
if os.getenv('VLLM_ATTENTION_BACKEND'):
|
||||
@@ -207,4 +331,9 @@ def get_engine_args():
|
||||
if os.getenv('DISABLE_LOG_REQUESTS', 'False').lower() == 'true':
|
||||
args['enable_log_requests'] = False
|
||||
|
||||
# Add speculative decoding configuration if present
|
||||
speculative_config = get_speculative_config()
|
||||
if speculative_config:
|
||||
args["speculative_config"] = speculative_config
|
||||
|
||||
return AsyncEngineArgs(**args)
|
||||
|
||||
+4
-4
@@ -6,7 +6,7 @@ from time import time
|
||||
|
||||
try:
|
||||
from vllm.utils import random_uuid
|
||||
from vllm.entrypoints.openai.engine.protocol import ErrorResponse, RequestResponseMetadata
|
||||
from vllm.entrypoints.openai.engine.protocol import ErrorResponse, ErrorInfo, RequestResponseMetadata
|
||||
from vllm import SamplingParams
|
||||
except ImportError:
|
||||
logging.warning("Error importing vllm, skipping related imports. This is ONLY expected when baking model into docker image from a machine without GPUs")
|
||||
@@ -87,9 +87,9 @@ class BatchSize:
|
||||
self.current_batch_size = min(self.current_batch_size*self.batch_size_growth_factor, self.max_batch_size)
|
||||
|
||||
def create_error_response(message: str, err_type: str = "BadRequestError", status_code: HTTPStatus = HTTPStatus.BAD_REQUEST) -> ErrorResponse:
|
||||
return ErrorResponse(message=message,
|
||||
type=err_type,
|
||||
code=status_code.value)
|
||||
return ErrorResponse(error=ErrorInfo(message=message,
|
||||
type=err_type,
|
||||
code=status_code.value))
|
||||
|
||||
def get_int_bool_env(env_var: str, default: bool) -> bool:
|
||||
return int(os.getenv(env_var, int(default))) == 1
|
||||
|
||||
Reference in New Issue
Block a user