diff --git a/Dockerfile b/Dockerfile index 988dc71..f8d16c2 100644 --- a/Dockerfile +++ b/Dockerfile @@ -12,7 +12,7 @@ RUN --mount=type=cache,target=/root/.cache/pip \ python3 -m pip install --upgrade -r /requirements.txt # Install vLLM (switching back to pip installs since issues that required building fork are fixed and space optimization is not as important since caching) and FlashInfer -RUN python3 -m pip install vllm==0.7.2 && \ +RUN python3 -m pip install vllm==0.7.3 && \ python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3 # Setup for Option 2: Building the Image with the Model included diff --git a/src/utils.py b/src/utils.py index 85af1e8..db6a95c 100644 --- a/src/utils.py +++ b/src/utils.py @@ -44,7 +44,7 @@ class JobInput: self.max_batch_size = job.get("max_batch_size") self.apply_chat_template = job.get("apply_chat_template", False) self.use_openai_format = job.get("use_openai_format", False) - self.sampling_params = SamplingParams(**job.get("sampling_params", {})) + self.sampling_params = SamplingParams(max_tokens=100, **job.get("sampling_params", {})) self.request_id = random_uuid() batch_size_growth_factor = job.get("batch_size_growth_factor") self.batch_size_growth_factor = float(batch_size_growth_factor) if batch_size_growth_factor else None