updating the handler

This commit is contained in:
Jorg Doku
2023-07-31 14:04:34 -05:00
parent b78a3d4429
commit 728e4f4f2e
2 changed files with 10 additions and 9 deletions
+1 -6
View File
@@ -1,10 +1,5 @@
# Required Python packages get listed here, one per line.
# Reccomended to lock the version number to avoid unexpected changes.
# Use the following additional index URL to get the latest PyTorch.
# --extra-index-url https://download.pytorch.org/whl/cu118
# vllm @ git+https://github.com/vllm-project/vllm.git@2b7d3aca2e1dd25fe26424f57c051af3b823cd71
# runpod @ git+https://github.com/runpod/runpod-python@vllm#egg=runpod
# Recomended to lock the version number to avoid unexpected changes.
vllm==0.1.2
huggingface-hub==0.16.4
runpod @ git+https://github.com/runpod/runpod-python@main#egg=runpod
+9 -3
View File
@@ -1,5 +1,5 @@
#!/usr/bin/env python
''' Contains the handler function that will be called by the serverless. '''
''' Contains the handler function that will be called by the serverless worker. '''
# Start the VLLM serving layer on our RunPod worker.
from templates import DEFAULT_TEMPLATE, LLAMA_TEMPLATE
@@ -9,8 +9,14 @@ import runpod
import os
# Prepare the model and tokenizer
MODEL = os.environ['MODEL']
TOKENIZER = os.environ['TOKENIZER']
MODEL = os.environ.get('MODEL', None)
TOKENIZER = os.environ.get('TOKENIZER', None)
if not MODEL:
print("Error: The model has not been provided.")
if not TOKENIZER:
print("Error: The tokenizer has not been provided.")
# Prepare the engine's arguments
engine_args = AsyncEngineArgs(