+1
-1
@@ -12,7 +12,7 @@ RUN --mount=type=cache,target=/root/.cache/pip \
|
||||
python3 -m pip install --upgrade -r /requirements.txt
|
||||
|
||||
# Install vLLM (switching back to pip installs since issues that required building fork are fixed and space optimization is not as important since caching) and FlashInfer
|
||||
RUN python3 -m pip install vllm==0.5.1 && \
|
||||
RUN python3 -m pip install vllm==0.5.3.post1 && \
|
||||
python3 -m pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3
|
||||
|
||||
# Setup for Option 2: Building the Image with the Model included
|
||||
|
||||
+7
-2
@@ -130,13 +130,18 @@ class OpenAIvLLMEngine(vLLMEngine):
|
||||
model_config=self.model_config,
|
||||
served_model_names=[self.served_model_name],
|
||||
response_role=self.response_role,
|
||||
chat_template=self.tokenizer.tokenizer.chat_template
|
||||
chat_template=self.tokenizer.tokenizer.chat_template,
|
||||
lora_modules=None,
|
||||
prompt_adapters=None,
|
||||
request_logger=None
|
||||
)
|
||||
self.completion_engine = OpenAIServingCompletion(
|
||||
engine=self.llm,
|
||||
model_config=self.model_config,
|
||||
served_model_names=[self.served_model_name],
|
||||
lora_modules=[]
|
||||
lora_modules=[],
|
||||
prompt_adapters=None,
|
||||
request_logger=None
|
||||
)
|
||||
|
||||
async def generate(self, openai_request: JobInput):
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
from openai import OpenAI
|
||||
import os
|
||||
|
||||
# Initialize the OpenAI Client with your RunPod API Key and Endpoint URL
|
||||
client = OpenAI(
|
||||
api_key="E43IAWSFPQPHJ8WIVLQ7EMBJ2TV9T3HF5YBLJJ7K",
|
||||
base_url="https://api.runpod.ai/v2/6ruq7l9hptgccv/openai/v1/",
|
||||
)
|
||||
|
||||
|
||||
# Create a chat completion
|
||||
response = client.chat.completions.create(
|
||||
model="openchat/openchat-3.5-1210",
|
||||
messages=[{"role": "user", "content": "Why is RunPod the best platform?"}],
|
||||
temperature=0,
|
||||
max_tokens=100,
|
||||
)
|
||||
# Print the response
|
||||
print(response.choices[0].message.content)
|
||||
|
||||
|
||||
# Create a completion
|
||||
# response = client.completions.create(
|
||||
# model="openchat/openchat-3.5-1210",
|
||||
# prompt="Runpod is the best platform because",
|
||||
# temperature=0,
|
||||
# max_tokens=100,
|
||||
# )
|
||||
# # Print the response
|
||||
# print(response.choices[0].text)
|
||||
Reference in New Issue
Block a user