diff --git a/Dockerfile b/Dockerfile index 0015fbe..0a08a18 100644 --- a/Dockerfile +++ b/Dockerfile @@ -12,7 +12,7 @@ RUN --mount=type=cache,target=/root/.cache/pip \ python3 -m pip install --upgrade -r /requirements.txt # Install vLLM (switching back to pip installs since issues that required building fork are fixed and space optimization is not as important since caching) and FlashInfer -RUN python3 -m pip install vllm==0.6.2 && \ +RUN python3 -m pip install vllm==0.6.3 && \ python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3 # Setup for Option 2: Building the Image with the Model included diff --git a/src/engine_args.py b/src/engine_args.py index 2f37696..45e50d1 100644 --- a/src/engine_args.py +++ b/src/engine_args.py @@ -88,7 +88,8 @@ DEFAULT_ARGS = { "typical_acceptance_sampler_posterior_alpha": float(os.getenv('TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA', 0)) or None, "qlora_adapter_name_or_path": os.getenv('QLORA_ADAPTER_NAME_OR_PATH', None), "disable_logprobs_during_spec_decoding": os.getenv('DISABLE_LOGPROBS_DURING_SPEC_DECODING', None), - "otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None) + "otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None), + "use_v2_block_manager": os.getenv('USE_V2_BLOCK_MANAGER', 'true') } def match_vllm_args(args):