update v0.5.3.post1

Signed-off-by: pandyamarut <pandyamarut@gmail.com>
This commit is contained in:
pandyamarut
2024-07-25 21:29:45 -07:00
parent 5bd6f3a75e
commit bd96b5e0de
3 changed files with 38 additions and 3 deletions
+1 -1
View File
@@ -12,7 +12,7 @@ RUN --mount=type=cache,target=/root/.cache/pip \
python3 -m pip install --upgrade -r /requirements.txt
# Install vLLM (switching back to pip installs since issues that required building fork are fixed and space optimization is not as important since caching) and FlashInfer
RUN python3 -m pip install vllm==0.5.1 && \
RUN python3 -m pip install vllm==0.5.3.post1 && \
python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3
# Setup for Option 2: Building the Image with the Model included
+7 -2
View File
@@ -130,13 +130,18 @@ class OpenAIvLLMEngine(vLLMEngine):
model_config=self.model_config,
served_model_names=[self.served_model_name],
response_role=self.response_role,
chat_template=self.tokenizer.tokenizer.chat_template
chat_template=self.tokenizer.tokenizer.chat_template,
lora_modules=None,
prompt_adapters=None,
request_logger=None
)
self.completion_engine = OpenAIServingCompletion(
engine=self.llm,
model_config=self.model_config,
served_model_names=[self.served_model_name],
lora_modules=[]
lora_modules=[],
prompt_adapters=None,
request_logger=None
)
async def generate(self, openai_request: JobInput):
+30
View File
@@ -0,0 +1,30 @@
from openai import OpenAI
import os
# Initialize the OpenAI Client with your RunPod API Key and Endpoint URL
client = OpenAI(
api_key="E43IAWSFPQPHJ8WIVLQ7EMBJ2TV9T3HF5YBLJJ7K",
base_url="https://api.runpod.ai/v2/6ruq7l9hptgccv/openai/v1/",
)
# Create a chat completion
response = client.chat.completions.create(
model="openchat/openchat-3.5-1210",
messages=[{"role": "user", "content": "Why is RunPod the best platform?"}],
temperature=0,
max_tokens=100,
)
# Print the response
print(response.choices[0].message.content)
# Create a completion
# response = client.completions.create(
# model="openchat/openchat-3.5-1210",
# prompt="Runpod is the best platform because",
# temperature=0,
# max_tokens=100,
# )
# # Print the response
# print(response.choices[0].text)