make minimal changes
This commit is contained in:
+17
-127
@@ -509,34 +509,13 @@
|
||||
"type": "string",
|
||||
"description": "Speculative decoding method to use.",
|
||||
"options": [
|
||||
{
|
||||
"label": "None",
|
||||
"value": ""
|
||||
},
|
||||
{
|
||||
"label": "Draft Model",
|
||||
"value": "draft_model"
|
||||
},
|
||||
{
|
||||
"label": "N-gram",
|
||||
"value": "ngram"
|
||||
},
|
||||
{
|
||||
"label": "EAGLE",
|
||||
"value": "eagle"
|
||||
},
|
||||
{
|
||||
"label": "EAGLE3",
|
||||
"value": "eagle3"
|
||||
},
|
||||
{
|
||||
"label": "Medusa",
|
||||
"value": "medusa"
|
||||
},
|
||||
{
|
||||
"label": "MLP Speculator",
|
||||
"value": "mlp_speculator"
|
||||
}
|
||||
{ "label": "None", "value": "" },
|
||||
{ "label": "Draft Model", "value": "draft_model" },
|
||||
{ "label": "N-gram", "value": "ngram" },
|
||||
{ "label": "EAGLE", "value": "eagle" },
|
||||
{ "label": "EAGLE3", "value": "eagle3" },
|
||||
{ "label": "Medusa", "value": "medusa" },
|
||||
{ "label": "MLP Speculator", "value": "mlp_speculator" }
|
||||
],
|
||||
"default": "",
|
||||
"advanced": true
|
||||
@@ -578,6 +557,16 @@
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "ENABLE_LOG_REQUESTS",
|
||||
"input": {
|
||||
"name": "Enable Log Requests",
|
||||
"type": "boolean",
|
||||
"description": "Enable vLLM request logging.",
|
||||
"default": false,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "TOKENIZER_NAME",
|
||||
"input": {
|
||||
@@ -655,35 +644,6 @@
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "ATTENTION_BACKEND",
|
||||
"input": {
|
||||
"name": "Attention Backend",
|
||||
"type": "string",
|
||||
"description": "Attention backend to use (e.g., FLASH_ATTN, XFORMERS, FLASHINFER).",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "ASYNC_SCHEDULING",
|
||||
"input": {
|
||||
"name": "Async Scheduling",
|
||||
"type": "boolean",
|
||||
"description": "Enable async scheduling for improved throughput.",
|
||||
"default": false,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "STREAM_INTERVAL",
|
||||
"input": {
|
||||
"name": "Stream Interval",
|
||||
"type": "number",
|
||||
"description": "Interval in seconds between streaming responses.",
|
||||
"default": 1,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "DEFAULT_BATCH_SIZE",
|
||||
"input": {
|
||||
@@ -844,76 +804,6 @@
|
||||
"default": "",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "TRUST_REQUEST_CHAT_TEMPLATE",
|
||||
"input": {
|
||||
"name": "Trust Request Chat Template",
|
||||
"type": "boolean",
|
||||
"description": "Allow chat templates from incoming requests to override the server default.",
|
||||
"default": false,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "RETURN_TOKENS_AS_TOKEN_IDS",
|
||||
"input": {
|
||||
"name": "Return Tokens as Token IDs",
|
||||
"type": "boolean",
|
||||
"description": "Return token IDs instead of token strings in responses.",
|
||||
"default": false,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "EXCLUDE_TOOLS_WHEN_TOOL_CHOICE_NONE",
|
||||
"input": {
|
||||
"name": "Exclude Tools When Tool Choice None",
|
||||
"type": "boolean",
|
||||
"description": "When tool_choice is 'none', exclude tool definitions from the prompt.",
|
||||
"default": false,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "ENABLE_PROMPT_TOKENS_DETAILS",
|
||||
"input": {
|
||||
"name": "Enable Prompt Tokens Details",
|
||||
"type": "boolean",
|
||||
"description": "Enable detailed prompt token usage in responses.",
|
||||
"default": false,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "ENABLE_FORCE_INCLUDE_USAGE",
|
||||
"input": {
|
||||
"name": "Enable Force Include Usage",
|
||||
"type": "boolean",
|
||||
"description": "Force include usage information in all streaming responses.",
|
||||
"default": false,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "ENABLE_LOG_OUTPUTS",
|
||||
"input": {
|
||||
"name": "Enable Log Outputs",
|
||||
"type": "boolean",
|
||||
"description": "Log model outputs for debugging.",
|
||||
"default": false,
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"key": "LOG_ERROR_STACK",
|
||||
"input": {
|
||||
"name": "Log Error Stack",
|
||||
"type": "boolean",
|
||||
"description": "Log full error stack traces.",
|
||||
"default": false,
|
||||
"advanced": true
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
@@ -8,7 +8,7 @@ typing-extensions>=4.8.0
|
||||
pydantic
|
||||
pydantic-settings
|
||||
hf-transfer
|
||||
transformers>=4.56.0,<5
|
||||
transformers>=4.57.0
|
||||
bitsandbytes>=0.45.0
|
||||
kernels
|
||||
torch-c-dlpack-ext
|
||||
|
||||
+12
-2
@@ -17,8 +17,11 @@ Complete guide to all environment variables and configuration options for worker
|
||||
| `HF_TOKEN` | - | `str` | Hugging Face token for private and gated models. |
|
||||
| `DTYPE` | 'auto' | ['auto', 'half', 'float16', 'bfloat16', 'float', 'float32'] | Data type for model weights and activations. |
|
||||
| `KV_CACHE_DTYPE` | 'auto' | ['auto', 'fp8'] | Data type for KV cache storage. |
|
||||
| `QUANTIZATION_PARAM_PATH` | None | `str` | Path to the JSON file containing the KV cache scaling factors. |
|
||||
| `MAX_MODEL_LEN` | None | `int` | Model context length. |
|
||||
| `GUIDED_DECODING_BACKEND` | 'outlines' | ['outlines', 'lm-format-enforcer'] | Which engine will be used for guided decoding by default. |
|
||||
| `DISTRIBUTED_EXECUTOR_BACKEND` | None | ['ray', 'mp'] | Backend to use for distributed serving. |
|
||||
| `WORKER_USE_RAY` | False | `bool` | Deprecated, use --distributed-executor-backend=ray. |
|
||||
| `PIPELINE_PARALLEL_SIZE` | 1 | `int` | Number of pipeline stages. |
|
||||
| `TENSOR_PARALLEL_SIZE` | 1 | `int` | Number of tensor parallel replicas. |
|
||||
| `MAX_PARALLEL_LOADING_WORKERS` | None | `int` | Load model sequentially in multiple batches. |
|
||||
@@ -33,6 +36,11 @@ Complete guide to all environment variables and configuration options for worker
|
||||
| `MAX_LOGPROBS` | 20 | `int` | Max number of log probs to return when logprobs is specified in SamplingParams. |
|
||||
| `DISABLE_LOG_STATS` | False | `bool` | Disable logging statistics. |
|
||||
| `QUANTIZATION` | None | ['awq', 'squeezellm', 'gptq', 'bitsandbytes'] | Method used to quantize the weights. |
|
||||
| `ROPE_SCALING` | None | `dict` | RoPE scaling configuration in JSON format. |
|
||||
| `ROPE_THETA` | None | `float` | RoPE theta. Use with rope_scaling. |
|
||||
| `TOKENIZER_POOL_SIZE` | 0 | `int` | Size of tokenizer pool to use for asynchronous tokenization. |
|
||||
| `TOKENIZER_POOL_TYPE` | 'ray' | `str` | Type of tokenizer pool to use for asynchronous tokenization. |
|
||||
| `TOKENIZER_POOL_EXTRA_CONFIG` | None | `dict` | Extra config for tokenizer pool. |
|
||||
|
||||
## LoRA (Low-Rank Adaptation) Settings
|
||||
|
||||
@@ -41,7 +49,9 @@ Complete guide to all environment variables and configuration options for worker
|
||||
| `ENABLE_LORA` | False | `bool` | If True, enable handling of LoRA adapters. |
|
||||
| `MAX_LORAS` | 1 | `int` | Max number of LoRAs in a single batch. |
|
||||
| `MAX_LORA_RANK` | 16 | `int` | Max LoRA rank. |
|
||||
| `LORA_EXTRA_VOCAB_SIZE` | 256 | `int` | Maximum size of extra vocabulary for LoRA adapters. |
|
||||
| `LORA_DTYPE` | 'auto' | ['auto', 'float16', 'bfloat16', 'float32'] | Data type for LoRA. |
|
||||
| `LONG_LORA_SCALING_FACTORS` | None | `tuple` | Specify multiple scaling factors for LoRA adapters. |
|
||||
| `MAX_CPU_LORAS` | None | `int` | Maximum number of LoRAs to store in CPU memory. |
|
||||
| `FULLY_SHARDED_LORAS` | False | `bool` | Enable fully sharded LoRA layers. |
|
||||
| `LORA_MODULES` | `[]` | `list[dict]` | Add lora adapters from Hugging Face `[{"name": "xx", "path": "xxx/xxxx", "base_model_name": "xxx/xxxx"}]` |
|
||||
@@ -153,11 +163,11 @@ These variables are used when building custom Docker images with models baked in
|
||||
| Variable | Default | Type | Description |
|
||||
| --------------------- | ---------------- | ----- | ------------------------------------------------- |
|
||||
| `BASE_PATH` | `/runpod-volume` | `str` | Storage directory for huggingface cache and model |
|
||||
| `WORKER_CUDA_VERSION` | `12.9.1` | `str` | CUDA version for the worker image |
|
||||
| `WORKER_CUDA_VERSION` | `12.1.0` | `str` | CUDA version for the worker image |
|
||||
|
||||
## Deprecated Variables
|
||||
|
||||
> **The following variables are deprecated and will be removed in future versions:**
|
||||
⚠️ **The following variables are deprecated and will be removed in future versions:**
|
||||
|
||||
| Old Variable | New Variable | Note |
|
||||
| ---------------------------- | ------------------------ | -------------------------------------------------------------------- |
|
||||
|
||||
+50
-13
@@ -33,14 +33,18 @@ DEFAULT_ARGS = {
|
||||
"quantization_param_path": os.getenv('QUANTIZATION_PARAM_PATH', None),
|
||||
"seed": int(os.getenv('SEED', 0)),
|
||||
"max_model_len": int(os.getenv('MAX_MODEL_LEN', 0)) or None,
|
||||
"worker_use_ray": os.getenv('WORKER_USE_RAY', 'False').lower() == 'true',
|
||||
"distributed_executor_backend": os.getenv('DISTRIBUTED_EXECUTOR_BACKEND', None),
|
||||
"max_parallel_loading_workers": int(os.getenv('MAX_PARALLEL_LOADING_WORKERS', 0)) or None,
|
||||
"block_size": int(os.getenv('BLOCK_SIZE', 16)),
|
||||
"enable_prefix_caching": os.getenv('ENABLE_PREFIX_CACHING', 'False').lower() == 'true',
|
||||
"disable_sliding_window": os.getenv('DISABLE_SLIDING_WINDOW', 'False').lower() == 'true',
|
||||
# attention_backend replaces deprecated VLLM_ATTENTION_BACKEND env var
|
||||
"attention_backend": os.getenv('ATTENTION_BACKEND', None),
|
||||
"async_scheduling": os.getenv('ASYNC_SCHEDULING', 'False').lower() == 'true',
|
||||
"stream_interval": float(os.getenv('STREAM_INTERVAL', 0)),
|
||||
# Enabled by default for improved throughput. Set to False to disable if experiencing issues
|
||||
"async_scheduling": None if os.getenv('ASYNC_SCHEDULING') is None else os.getenv('ASYNC_SCHEDULING', 'True').lower() == 'true',
|
||||
# Controls how often to yield streaming results
|
||||
"stream_interval": int(os.getenv('STREAM_INTERVAL', 1)),
|
||||
"swap_space": int(os.getenv('SWAP_SPACE', 4)), # GiB
|
||||
"cpu_offload_gb": int(os.getenv('CPU_OFFLOAD_GB', 0)), # GiB
|
||||
# vLLM defaults None to 2048; keep 0 as None to let vLLM auto-calculate
|
||||
@@ -49,12 +53,17 @@ DEFAULT_ARGS = {
|
||||
"max_logprobs": int(os.getenv('MAX_LOGPROBS', 20)), # Default value for OpenAI Chat Completions API
|
||||
"revision": os.getenv('REVISION', None),
|
||||
"code_revision": os.getenv('CODE_REVISION', None),
|
||||
"rope_scaling": os.getenv('ROPE_SCALING', None),
|
||||
"rope_theta": float(os.getenv('ROPE_THETA', 0)) or None,
|
||||
"tokenizer_revision": os.getenv('TOKENIZER_REVISION', None),
|
||||
"quantization": os.getenv('QUANTIZATION', None),
|
||||
"enforce_eager": os.getenv('ENFORCE_EAGER', 'False').lower() == 'true',
|
||||
"max_context_len_to_capture": int(os.getenv('MAX_CONTEXT_LEN_TO_CAPTURE', 0)) or None,
|
||||
"max_seq_len_to_capture": int(os.getenv('MAX_SEQ_LEN_TO_CAPTURE', 8192)),
|
||||
"disable_custom_all_reduce": os.getenv('DISABLE_CUSTOM_ALL_REDUCE', 'False').lower() == 'true',
|
||||
"tokenizer_pool_size": int(os.getenv('TOKENIZER_POOL_SIZE', 0)),
|
||||
"tokenizer_pool_type": os.getenv('TOKENIZER_POOL_TYPE', 'ray'),
|
||||
"tokenizer_pool_extra_config": os.getenv('TOKENIZER_POOL_EXTRA_CONFIG', None),
|
||||
"enable_lora": os.getenv('ENABLE_LORA', 'False').lower() == 'true',
|
||||
"max_loras": int(os.getenv('MAX_LORAS', 1)),
|
||||
"max_lora_rank": int(os.getenv('MAX_LORA_RANK', 16)),
|
||||
@@ -62,19 +71,33 @@ DEFAULT_ARGS = {
|
||||
"max_prompt_adapters": int(os.getenv('MAX_PROMPT_ADAPTERS', 1)),
|
||||
"max_prompt_adapter_token": int(os.getenv('MAX_PROMPT_ADAPTER_TOKEN', 0)),
|
||||
"fully_sharded_loras": os.getenv('FULLY_SHARDED_LORAS', 'False').lower() == 'true',
|
||||
"lora_extra_vocab_size": int(os.getenv('LORA_EXTRA_VOCAB_SIZE', 256)),
|
||||
"long_lora_scaling_factors": tuple(map(float, os.getenv('LONG_LORA_SCALING_FACTORS', '').split(','))) if os.getenv('LONG_LORA_SCALING_FACTORS') else None,
|
||||
"lora_dtype": os.getenv('LORA_DTYPE', 'auto'),
|
||||
"max_cpu_loras": int(os.getenv('MAX_CPU_LORAS', 0)) or None,
|
||||
"device": os.getenv('DEVICE', 'auto'),
|
||||
"ray_workers_use_nsight": os.getenv('RAY_WORKERS_USE_NSIGHT', 'False').lower() == 'true',
|
||||
"num_gpu_blocks_override": int(os.getenv('NUM_GPU_BLOCKS_OVERRIDE', 0)) or None,
|
||||
"num_lookahead_slots": int(os.getenv('NUM_LOOKAHEAD_SLOTS', 0)),
|
||||
"model_loader_extra_config": os.getenv('MODEL_LOADER_EXTRA_CONFIG', None),
|
||||
"ignore_patterns": os.getenv('IGNORE_PATTERNS', None),
|
||||
"preemption_mode": os.getenv('PREEMPTION_MODE', None),
|
||||
"scheduler_delay_factor": float(os.getenv('SCHEDULER_DELAY_FACTOR', 0.0)),
|
||||
"enable_chunked_prefill": os.getenv('ENABLE_CHUNKED_PREFILL', None),
|
||||
"guided_decoding_backend": os.getenv('GUIDED_DECODING_BACKEND', 'outlines'),
|
||||
"speculative_model": os.getenv('SPECULATIVE_MODEL', None),
|
||||
"speculative_draft_tensor_parallel_size": int(os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE', 0)) or None,
|
||||
"enable_expert_parallel": bool(os.getenv('ENABLE_EXPERT_PARALLEL', 'False').lower() == 'true'),
|
||||
"num_speculative_tokens": int(os.getenv('NUM_SPECULATIVE_TOKENS', 0)) or None,
|
||||
"speculative_max_model_len": int(os.getenv('SPECULATIVE_MAX_MODEL_LEN', 0)) or None,
|
||||
"speculative_disable_by_batch_size": int(os.getenv('SPECULATIVE_DISABLE_BY_BATCH_SIZE', 0)) or None,
|
||||
"ngram_prompt_lookup_max": int(os.getenv('NGRAM_PROMPT_LOOKUP_MAX', 0)) or None,
|
||||
"ngram_prompt_lookup_min": int(os.getenv('NGRAM_PROMPT_LOOKUP_MIN', 0)) or None,
|
||||
"spec_decoding_acceptance_method": os.getenv('SPEC_DECODING_ACCEPTANCE_METHOD', 'rejection_sampler'),
|
||||
"typical_acceptance_sampler_posterior_threshold": float(os.getenv('TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD', 0)) or None,
|
||||
"typical_acceptance_sampler_posterior_alpha": float(os.getenv('TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA', 0)) or None,
|
||||
"qlora_adapter_name_or_path": os.getenv('QLORA_ADAPTER_NAME_OR_PATH', None),
|
||||
"disable_logprobs_during_spec_decoding": os.getenv('DISABLE_LOGPROBS_DURING_SPEC_DECODING', None),
|
||||
"otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None),
|
||||
}
|
||||
|
||||
@@ -241,20 +264,34 @@ def get_engine_args():
|
||||
args["max_seq_len_to_capture"] = int(os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE"))
|
||||
logging.warning("Using MAX_CONTEXT_LEN_TO_CAPTURE is deprecated. Please use MAX_SEQ_LEN_TO_CAPTURE instead.")
|
||||
|
||||
# VLLM_ATTENTION_BACKEND env var → attention_backend arg (deprecated shim)
|
||||
vllm_attn_backend = os.getenv("VLLM_ATTENTION_BACKEND")
|
||||
if vllm_attn_backend and "attention_backend" not in args:
|
||||
args["attention_backend"] = vllm_attn_backend
|
||||
logging.warning("VLLM_ATTENTION_BACKEND is deprecated. Please use ATTENTION_BACKEND instead.")
|
||||
# if "gemma-2" in args.get("model", "").lower():
|
||||
# os.environ["VLLM_ATTENTION_BACKEND"] = "FLASHINFER"
|
||||
# logging.info("Using FLASHINFER for gemma-2 model.")
|
||||
|
||||
# DISABLE_LOG_REQUESTS → enable_log_requests (inverted, deprecated shim)
|
||||
if os.getenv("DISABLE_LOG_REQUESTS") and "enable_log_requests" not in args:
|
||||
args["enable_log_requests"] = os.getenv("DISABLE_LOG_REQUESTS", "False").lower() != "true"
|
||||
logging.warning("DISABLE_LOG_REQUESTS is deprecated. Please use ENABLE_LOG_REQUESTS instead.")
|
||||
|
||||
# Default max_num_batched_tokens to max_model_len when not explicitly set
|
||||
# When max_num_batched_tokens is None (env var was 0), set to max_model_len
|
||||
# to preserve "unlimited" behavior. vLLM defaults None to 2048.
|
||||
if args.get("max_num_batched_tokens") is None and args.get("max_model_len") is not None:
|
||||
args["max_num_batched_tokens"] = args["max_model_len"]
|
||||
logging.info(f"Setting max_num_batched_tokens to max_model_len ({args['max_model_len']}) for unlimited batching.")
|
||||
|
||||
# VLLM_ATTENTION_BACKEND is deprecated, migrate to attention_backend
|
||||
if os.getenv('VLLM_ATTENTION_BACKEND'):
|
||||
logging.warning(
|
||||
"VLLM_ATTENTION_BACKEND env var is deprecated. "
|
||||
"Use ATTENTION_BACKEND instead (maps to --attention-backend CLI arg)."
|
||||
)
|
||||
if not args.get('attention_backend'):
|
||||
args['attention_backend'] = os.getenv('VLLM_ATTENTION_BACKEND')
|
||||
|
||||
# DISABLE_LOG_REQUESTS is deprecated, use ENABLE_LOG_REQUESTS instead
|
||||
if os.getenv('DISABLE_LOG_REQUESTS'):
|
||||
logging.warning(
|
||||
"DISABLE_LOG_REQUESTS env var is deprecated. "
|
||||
"Use ENABLE_LOG_REQUESTS instead (default: False)."
|
||||
)
|
||||
# Honor old behavior: if DISABLE_LOG_REQUESTS=true, don't enable logging
|
||||
if os.getenv('DISABLE_LOG_REQUESTS', 'False').lower() == 'true':
|
||||
args['enable_log_requests'] = False
|
||||
|
||||
# Add speculative decoding configuration if present
|
||||
speculative_config = get_speculative_config()
|
||||
|
||||
Reference in New Issue
Block a user