diff --git a/Dockerfile b/Dockerfile index 426d771..089b4de 100644 --- a/Dockerfile +++ b/Dockerfile @@ -12,7 +12,7 @@ RUN --mount=type=cache,target=/root/.cache/pip \ python3 -m pip install --upgrade -r /requirements.txt # Install vLLM (switching back to pip installs since issues that required building fork are fixed and space optimization is not as important since caching) and FlashInfer -RUN python3 -m pip install vllm==0.5.1 && \ +RUN python3 -m pip install vllm==0.5.3.post1 && \ python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3 # Setup for Option 2: Building the Image with the Model included diff --git a/src/engine.py b/src/engine.py index 33b374b..1f48145 100644 --- a/src/engine.py +++ b/src/engine.py @@ -130,13 +130,18 @@ class OpenAIvLLMEngine(vLLMEngine): model_config=self.model_config, served_model_names=[self.served_model_name], response_role=self.response_role, - chat_template=self.tokenizer.tokenizer.chat_template + chat_template=self.tokenizer.tokenizer.chat_template, + lora_modules=None, + prompt_adapters=None, + request_logger=None ) self.completion_engine = OpenAIServingCompletion( engine=self.llm, model_config=self.model_config, served_model_names=[self.served_model_name], - lora_modules=[] + lora_modules=[], + prompt_adapters=None, + request_logger=None ) async def generate(self, openai_request: JobInput): diff --git a/test.py b/test.py new file mode 100644 index 0000000..76e0b83 --- /dev/null +++ b/test.py @@ -0,0 +1,30 @@ +from openai import OpenAI +import os + +# Initialize the OpenAI Client with your RunPod API Key and Endpoint URL +client = OpenAI( + api_key="E43IAWSFPQPHJ8WIVLQ7EMBJ2TV9T3HF5YBLJJ7K", + base_url="https://api.runpod.ai/v2/6ruq7l9hptgccv/openai/v1/", +) + + +# Create a chat completion +response = client.chat.completions.create( + model="openchat/openchat-3.5-1210", + messages=[{"role": "user", "content": "Why is RunPod the best platform?"}], + temperature=0, + max_tokens=100, +) +# Print the response +print(response.choices[0].message.content) + + +# Create a completion +# response = client.completions.create( +# model="openchat/openchat-3.5-1210", +# prompt="Runpod is the best platform because", +# temperature=0, +# max_tokens=100, +# ) +# # Print the response +# print(response.choices[0].text) \ No newline at end of file