remove DEFAULT_ARGS that are none, fix MAX_CONTEXT_LEN_TO_CAPTURE

This commit is contained in:
velaraptor-runpod
2026-02-27 14:04:15 -06:00
parent cd485a1af1
commit 8a9365bed4
-35
View File
@@ -12,7 +12,6 @@ ENV_ALIASES = {
"MODEL_NAME": "model",
"MODEL_REVISION": "revision",
"TOKENIZER_NAME": "tokenizer",
"MAX_CONTEXT_LEN_TO_CAPTURE": "max_seq_len_to_capture",
}
# Literal defaults from original worker (used when env/local do not set a value)
@@ -22,41 +21,26 @@ DEFAULT_ARGS = {
"gpu_memory_utilization": 0.95,
"pipeline_parallel_size": 1,
"tensor_parallel_size": 1,
"served_model_name": None,
"tokenizer": None,
"skip_tokenizer_init": False,
"tokenizer_mode": "auto",
"trust_remote_code": False,
"download_dir": None,
"load_format": "auto",
"dtype": "auto",
"kv_cache_dtype": "auto",
"quantization_param_path": None,
"seed": 0,
"max_model_len": None,
"worker_use_ray": False,
"distributed_executor_backend": None,
"max_parallel_loading_workers": None,
"block_size": 16,
"enable_prefix_caching": False,
"disable_sliding_window": False,
"swap_space": 4,
"cpu_offload_gb": 0,
"max_num_batched_tokens": None,
"max_num_seqs": 256,
"max_logprobs": 20,
"revision": None,
"code_revision": None,
"rope_scaling": None,
"rope_theta": None,
"tokenizer_revision": None,
"quantization": None,
"enforce_eager": False,
"max_seq_len_to_capture": 8192,
"disable_custom_all_reduce": False,
"tokenizer_pool_size": 0,
"tokenizer_pool_type": "ray",
"tokenizer_pool_extra_config": None,
"enable_lora": False,
"max_loras": 1,
"max_lora_rank": 16,
@@ -65,32 +49,13 @@ DEFAULT_ARGS = {
"max_prompt_adapter_token": 0,
"fully_sharded_loras": False,
"lora_extra_vocab_size": 256,
"long_lora_scaling_factors": None,
"lora_dtype": "auto",
"max_cpu_loras": None,
"device": "auto",
"ray_workers_use_nsight": False,
"num_gpu_blocks_override": None,
"num_lookahead_slots": 0,
"model_loader_extra_config": None,
"ignore_patterns": None,
"preemption_mode": None,
"scheduler_delay_factor": 0.0,
"enable_chunked_prefill": None,
"guided_decoding_backend": "outlines",
"speculative_model": None,
"speculative_draft_tensor_parallel_size": None,
"num_speculative_tokens": None,
"speculative_max_model_len": None,
"speculative_disable_by_batch_size": None,
"ngram_prompt_lookup_max": None,
"ngram_prompt_lookup_min": None,
"spec_decoding_acceptance_method": "rejection_sampler",
"typical_acceptance_sampler_posterior_threshold": None,
"typical_acceptance_sampler_posterior_alpha": None,
"qlora_adapter_name_or_path": None,
"disable_logprobs_during_spec_decoding": None,
"otlp_traces_endpoint": None,
"stream_interval": 1,
}