Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
69646b9e99 | ||
|
|
8b991a7ad7 | ||
|
|
dac05b62b3 | ||
|
|
d356c31675 | ||
|
|
14b74a4989 | ||
|
|
80072047ab |
@@ -33,6 +33,17 @@ All behaviour is controlled through environment variables:
|
|||||||
|
|
||||||
**Pass any vLLM engine arg** not listed above by setting an env var with the **UPPERCASED** field name (e.g. `MAX_MODEL_LEN=4096`, `ENABLE_CHUNKED_PREFILL=true`). The worker auto-discovers all `AsyncEngineArgs` fields from env. See the [vLLM engine args docs](https://docs.vllm.ai/en/latest/configuration/engine_args) for all available options.
|
**Pass any vLLM engine arg** not listed above by setting an env var with the **UPPERCASED** field name (e.g. `MAX_MODEL_LEN=4096`, `ENABLE_CHUNKED_PREFILL=true`). The worker auto-discovers all `AsyncEngineArgs` fields from env. See the [vLLM engine args docs](https://docs.vllm.ai/en/latest/configuration/engine_args) for all available options.
|
||||||
|
|
||||||
|
**Configuration file:** You can also supply a `config.yaml` instead of (or alongside) env vars. Mount it at `/vllm_config.yaml` in the container, or set `VLLM_CONFIG_FILE` to a custom path. Use the same key names as `vllm serve` — hyphens and underscores both work:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model: meta-llama/Llama-3.1-8B-Instruct
|
||||||
|
max-model-len: 8192
|
||||||
|
gpu-memory-utilization: 0.90
|
||||||
|
quantization: awq
|
||||||
|
```
|
||||||
|
|
||||||
|
Environment variables always override config file values.
|
||||||
|
|
||||||
For complete configuration options, see the [full configuration documentation](https://github.com/runpod-workers/worker-vllm/blob/main/docs/configuration.md).
|
For complete configuration options, see the [full configuration documentation](https://github.com/runpod-workers/worker-vllm/blob/main/docs/configuration.md).
|
||||||
|
|
||||||
### Specify Transformers Version
|
### Specify Transformers Version
|
||||||
|
|||||||
@@ -5,7 +5,7 @@
|
|||||||
"input": {
|
"input": {
|
||||||
"prompt": "Write a short poem about artificial intelligence."
|
"prompt": "Write a short poem about artificial intelligence."
|
||||||
},
|
},
|
||||||
"timeout": 30000
|
"timeout": 300000
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"name": "openai_messages_test",
|
"name": "openai_messages_test",
|
||||||
@@ -26,7 +26,7 @@
|
|||||||
"temperature": 0.1
|
"temperature": 0.1
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
"timeout": 30000
|
"timeout": 300000
|
||||||
}
|
}
|
||||||
],
|
],
|
||||||
"config": {
|
"config": {
|
||||||
@@ -38,6 +38,6 @@
|
|||||||
"value": "HuggingFaceTB/SmolLM2-135M-Instruct"
|
"value": "HuggingFaceTB/SmolLM2-135M-Instruct"
|
||||||
}
|
}
|
||||||
],
|
],
|
||||||
"allowedCudaVersions": ["12.9", "12.8", "12.7", "12.6", "12.5"]
|
"allowedCudaVersions": ["13.0"]
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -78,6 +78,20 @@ Configure worker-vllm using environment variables:
|
|||||||
|
|
||||||
Any env var whose name matches a valid `AsyncEngineArgs` field (uppercased) is applied automatically. Backward-compat aliases: `MODEL_NAME`, `TOKENIZER_NAME`, `MAX_CONTEXT_LEN_TO_CAPTURE`. This lets you configure any vLLM option without waiting for explicit worker support.
|
Any env var whose name matches a valid `AsyncEngineArgs` field (uppercased) is applied automatically. Backward-compat aliases: `MODEL_NAME`, `TOKENIZER_NAME`, `MAX_CONTEXT_LEN_TO_CAPTURE`. This lets you configure any vLLM option without waiting for explicit worker support.
|
||||||
|
|
||||||
|
### Configuration File (config.yaml)
|
||||||
|
|
||||||
|
As an alternative to environment variables, you can supply a `config.yaml` file using the same key names as `vllm serve` (hyphens or underscores both work):
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
model: meta-llama/Llama-3.1-8B-Instruct
|
||||||
|
max-model-len: 8192
|
||||||
|
gpu-memory-utilization: 0.90
|
||||||
|
quantization: awq
|
||||||
|
tensor-parallel-size: 2
|
||||||
|
```
|
||||||
|
|
||||||
|
Mount the file into the container at `/vllm_config.yaml`, or point to a custom path with the `VLLM_CONFIG_FILE` env var. Environment variables always take precedence over config file values.
|
||||||
|
|
||||||
For the complete list of all available environment variables, examples, and detailed descriptions: **[Configuration](docs/configuration.md)**
|
For the complete list of all available environment variables, examples, and detailed descriptions: **[Configuration](docs/configuration.md)**
|
||||||
|
|
||||||
### Specify Transformers Version
|
### Specify Transformers Version
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
ray
|
ray
|
||||||
pandas
|
pandas
|
||||||
pyarrow
|
pyarrow
|
||||||
runpod==1.9.0
|
runpod==1.9.1
|
||||||
huggingface-hub
|
huggingface-hub
|
||||||
lmcache==0.4.5
|
lmcache==0.4.5
|
||||||
packaging>=24.2
|
packaging>=24.2
|
||||||
@@ -11,5 +11,5 @@ pydantic-settings
|
|||||||
hf-transfer
|
hf-transfer
|
||||||
transformers>=5
|
transformers>=5
|
||||||
bitsandbytes>=0.45.0
|
bitsandbytes>=0.45.0
|
||||||
kernels
|
kernels<0.15
|
||||||
torch-c-dlpack-ext
|
torch-c-dlpack-ext
|
||||||
|
|||||||
@@ -404,6 +404,23 @@ def _resolve_cached_model_path(model_name: str) -> str:
|
|||||||
return resolved
|
return resolved
|
||||||
|
|
||||||
|
|
||||||
|
def _get_args_from_config_file() -> dict:
|
||||||
|
"""Load engine args from a vLLM-style config.yaml.
|
||||||
|
|
||||||
|
Checks VLLM_CONFIG_FILE env var, then falls back to /vllm_config.yaml.
|
||||||
|
Keys use the same long-form names as vllm serve (hyphens converted to underscores).
|
||||||
|
"""
|
||||||
|
import yaml
|
||||||
|
path = os.getenv("VLLM_CONFIG_FILE", "/vllm_config.yaml")
|
||||||
|
if not os.path.exists(path):
|
||||||
|
return {}
|
||||||
|
with open(path) as f:
|
||||||
|
raw = yaml.safe_load(f) or {}
|
||||||
|
normalized = {k.replace("-", "_"): v for k, v in raw.items()}
|
||||||
|
logging.info("Loaded engine args from config file %s: %s", path, list(normalized.keys()))
|
||||||
|
return normalized
|
||||||
|
|
||||||
|
|
||||||
def get_local_args():
|
def get_local_args():
|
||||||
"""
|
"""
|
||||||
Retrieve local arguments from a JSON file.
|
Retrieve local arguments from a JSON file.
|
||||||
@@ -429,6 +446,9 @@ def get_engine_args():
|
|||||||
# Start with worker custom defaults (only where we differ from vLLM)
|
# Start with worker custom defaults (only where we differ from vLLM)
|
||||||
args = dict(DEFAULT_ARGS)
|
args = dict(DEFAULT_ARGS)
|
||||||
|
|
||||||
|
# Config file values sit above defaults but below env vars
|
||||||
|
args.update(_get_args_from_config_file())
|
||||||
|
|
||||||
# Auto-discover: every AsyncEngineArgs field from env UPPERCASED (e.g. MAX_MODEL_LEN)
|
# Auto-discover: every AsyncEngineArgs field from env UPPERCASED (e.g. MAX_MODEL_LEN)
|
||||||
args.update(_get_args_from_env_auto_discover())
|
args.update(_get_args_from_env_auto_discover())
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user