From f5a063956e093238329371bfefc35b80dced0adb Mon Sep 17 00:00:00 2001 From: Jhenner Tigreros <32320832+JhennerTigreros@users.noreply.github.com> Date: Thu, 7 Aug 2025 15:13:37 -0500 Subject: [PATCH 1/4] Fix requirements.txt to support gpt-oss models --- builder/requirements.txt | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/builder/requirements.txt b/builder/requirements.txt index 028853f..1029d7c 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -8,5 +8,7 @@ typing-extensions>=4.8.0 pydantic pydantic-settings hf-transfer -transformers +transformers>=4.55.0 bitsandbytes>=0.45.0 +kernels +torch==2.6.0 From d8863139d611377ff0e4d8b1fc6c70016c53978d Mon Sep 17 00:00:00 2001 From: Jhenner Tigreros Date: Thu, 7 Aug 2025 15:28:02 -0500 Subject: [PATCH 2/4] add model to test --- Dockerfile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Dockerfile b/Dockerfile index a2ce849..8b8ef9e 100644 --- a/Dockerfile +++ b/Dockerfile @@ -16,7 +16,7 @@ RUN python3 -m pip install vllm==0.9.1 && \ python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3 # Setup for Option 2: Building the Image with the Model included -ARG MODEL_NAME="" +ARG MODEL_NAME="openai/gpt-oss-120b" ARG TOKENIZER_NAME="" ARG BASE_PATH="/runpod-volume" ARG QUANTIZATION="" From 8b02a703b4cba58f56c34b65be6d876702c4dcd7 Mon Sep 17 00:00:00 2001 From: Jhenner Tigreros Date: Thu, 7 Aug 2025 15:59:43 -0500 Subject: [PATCH 3/4] fix issues --- Dockerfile | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Dockerfile b/Dockerfile index 8b8ef9e..bb22ceb 100644 --- a/Dockerfile +++ b/Dockerfile @@ -12,11 +12,11 @@ RUN --mount=type=cache,target=/root/.cache/pip \ python3 -m pip install --upgrade -r /requirements.txt # Install vLLM (switching back to pip installs since issues that required building fork are fixed and space optimization is not as important since caching) and FlashInfer -RUN python3 -m pip install vllm==0.9.1 && \ +RUN python3 -m pip install vllm==0.10.0 && \ python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3 # Setup for Option 2: Building the Image with the Model included -ARG MODEL_NAME="openai/gpt-oss-120b" +ARG MODEL_NAME="" ARG TOKENIZER_NAME="" ARG BASE_PATH="/runpod-volume" ARG QUANTIZATION="" From fb0c0307971313ae5a5ada59701eafdc2ecd33c6 Mon Sep 17 00:00:00 2001 From: Jhenner Tigreros Date: Thu, 7 Aug 2025 16:40:06 -0500 Subject: [PATCH 4/4] fix initialization on openaiservingmodels --- src/engine.py | 1 - 1 file changed, 1 deletion(-) diff --git a/src/engine.py b/src/engine.py index 00cf401..073f0d5 100644 --- a/src/engine.py +++ b/src/engine.py @@ -207,7 +207,6 @@ class OpenAIvLLMEngine(vLLMEngine): model_config=self.model_config, base_model_paths=self.base_model_paths, lora_modules=self.lora_adapters, - prompt_adapters=None, ) await self.serving_models.init_static_loras()