From efb093e1988bf0f0ad20b31e1010d2ea4e28324f Mon Sep 17 00:00:00 2001 From: velaraptor-runpod Date: Tue, 24 Feb 2026 17:37:57 -0600 Subject: [PATCH] add as VLLM_RUNPOD prefix and update readme --- .runpod/README.md | 2 ++ README.md | 10 ++++++++++ docs/configuration.md | 22 ++++++++++++++++++++++ src/engine_args.py | 14 +++++++------- 4 files changed, 41 insertions(+), 7 deletions(-) diff --git a/.runpod/README.md b/.runpod/README.md index f0f4b04..d5d1138 100644 --- a/.runpod/README.md +++ b/.runpod/README.md @@ -28,6 +28,8 @@ All behaviour is controlled through environment variables: | `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | Override served model name in API | | String | | `MAX_CONCURRENCY` | Maximum concurrent requests | 300 | Integer | +**Pass any vLLM engine arg** not listed above by prefixing it with `VLLM_RUNPOD_`. The suffix maps to the vLLM `AsyncEngineArgs` field name (case-insensitive). For example, `VLLM_RUNPOD_ENABLE_CHUNKED_PREFILL=true` sets `enable_chunked_prefill`. See the [vLLM engine args docs](https://docs.vllm.ai/en/latest/configuration/engine_args) for all available options. + For complete configuration options, see the [full configuration documentation](https://github.com/runpod-workers/worker-vllm/blob/main/docs/configuration.md). ## API Usage diff --git a/README.md b/README.md index c64180e..f3eed91 100644 --- a/README.md +++ b/README.md @@ -59,6 +59,16 @@ Configure worker-vllm using environment variables: | `OPENAI_SERVED_MODEL_NAME_OVERRIDE` | Override served model name in API | | String | | `MAX_CONCURRENCY` | Maximum concurrent requests | 30 | Integer | +**Pass any vLLM engine arg** not listed above by prefixing it with `VLLM_`. The suffix maps directly to the vLLM `AsyncEngineArgs` field name (case-insensitive). For example: + +| Environment Variable | vLLM Engine Arg | Example Value | +| ------------------------- | ------------------------ | ------------- | +| `VLLM_RUNPOD_MAX_MODEL_LEN` | `max_model_len` | `4096` | +| `VLLM_RUNPOD_ENFORCE_EAGER` | `enforce_eager` | `true` | +| `VLLM_RUNPOD_ENABLE_CHUNKED_PREFILL` | `enable_chunked_prefill` | `true` | + +Any `VLLM_RUNPOD_` that matches a valid vLLM engine arg will be applied automatically. This lets you configure any vLLM option without waiting for explicit worker support. + For the complete list of all available environment variables, examples, and detailed descriptions: **[Configuration](docs/configuration.md)** ## Option 2: Build Docker Image with Model Inside diff --git a/docs/configuration.md b/docs/configuration.md index 323fce3..f675a51 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -156,6 +156,28 @@ The way this works is that the first request will have a batch size of `DEFAULT_ | `DISABLE_LOGGING_REQUEST` | False | `bool` | Disable logging requests. | | `MAX_LOG_LEN` | None | `int` | Max number of prompt characters or prompt ID numbers being printed in log. | +## VLLM_RUNPOD_ Prefix: Pass Any Engine Arg + +Any vLLM `AsyncEngineArgs` field can be set via an environment variable using the `VLLM_RUNPOD_` prefix. The suffix maps directly to the field name (case-insensitive, underscores preserved). + +**Format:** `VLLM_RUNPOD_=` + +**Examples:** + +| Environment Variable | vLLM Engine Arg | Value Example | +| ------------------------------------- | -------------------------- | ------------- | +| `VLLM_RUNPOD_MAX_MODEL_LEN` | `max_model_len` | `4096` | +| `VLLM_RUNPOD_ENFORCE_EAGER` | `enforce_eager` | `true` | +| `VLLM_RUNPOD_ENABLE_CHUNKED_PREFILL` | `enable_chunked_prefill` | `true` | +| `VLLM_RUNPOD_NUM_SCHEDULER_STEPS` | `num_scheduler_steps` | `8` | +| `VLLM_RUNPOD_TOKENIZER_POOL_SIZE` | `tokenizer_pool_size` | `4` | + +**Notes:** +- Only valid `AsyncEngineArgs` fields are applied. Unknown keys are silently ignored. +- Values are automatically cast to the correct type (`int`, `float`, `bool`, `str`, or JSON for `dict`/`list`). +- `VLLM_RUNPOD_` overrides are applied **after** all other worker env vars, so they take precedence. +- For a full list of available engine args, see the [vLLM AsyncEngineArgs documentation](https://docs.vllm.ai/en/latest/serving/engine_args.html). + ## Docker Build Arguments These variables are used when building custom Docker images with models baked in: diff --git a/src/engine_args.py b/src/engine_args.py index fa12f52..ecdab9e 100644 --- a/src/engine_args.py +++ b/src/engine_args.py @@ -14,7 +14,7 @@ RENAME_ARGS_MAP = { "MAX_CONTEXT_LEN_TO_CAPTURE": "max_seq_len_to_capture" } -VLLM_ENV_PREFIX = "VLLM_" +VLLM_ENV_PREFIX = "VLLM_RUNPOD_" def _resolve_field_type(field_type: type) -> type: @@ -70,10 +70,10 @@ def _convert_env_value_to_field_type(value: str, field_name: str, field_type: ty def _get_vllm_env_overrides() -> dict: - """Collect engine arg overrides from env vars with prefix VLLM_. + """Collect engine arg overrides from env vars with prefix VLLM_RUNPOD_. - Any env var VLLM_ maps to the engine arg (lowercase). - E.g. VLLM_MAX_MODEL_LEN=4096 -> max_model_len=4096. + Any env var VLLM_RUNPOD_ maps to the engine arg (lowercase). + E.g. VLLM_RUNPOD_MAX_MODEL_LEN=4096 -> max_model_len=4096. Only keys that exist on AsyncEngineArgs are applied; values are converted to the field type (int, float, bool, str, json for dict/list). """ @@ -93,12 +93,12 @@ def _get_vllm_env_overrides() -> dict: ) except (ValueError, TypeError, json.JSONDecodeError) as e: logging.warning( - "Skip VLLM_ env override %s=%r: %s", key, value, e + "Skip VLLM_RUNPOD_ env override %s=%r: %s", key, value, e ) continue if overrides: logging.info( - "Applying engine arg overrides from VLLM_ env vars: %s", + "Applying engine arg overrides from VLLM_RUNPOD_ env vars: %s", list(overrides.keys()), ) return overrides @@ -359,7 +359,7 @@ def get_engine_args(): # Rename and match to vllm args args = match_vllm_args(args) - # Apply any VLLM_* env vars as overrides (e.g. VLLM_MAX_MODEL_LEN=4096 -> max_model_len=4096) + # Apply any VLLM_RUNPOD_* env vars as overrides (e.g. VLLM_RUNPOD_MAX_MODEL_LEN=4096 -> max_model_len=4096) args.update(_get_vllm_env_overrides()) if args.get("load_format") == "bitsandbytes":