updating the handler
This commit is contained in:
@@ -1,10 +1,5 @@
|
||||
# Required Python packages get listed here, one per line.
|
||||
# Reccomended to lock the version number to avoid unexpected changes.
|
||||
|
||||
# Use the following additional index URL to get the latest PyTorch.
|
||||
# --extra-index-url https://download.pytorch.org/whl/cu118
|
||||
# vllm @ git+https://github.com/vllm-project/vllm.git@2b7d3aca2e1dd25fe26424f57c051af3b823cd71
|
||||
# runpod @ git+https://github.com/runpod/runpod-python@vllm#egg=runpod
|
||||
# Recomended to lock the version number to avoid unexpected changes.
|
||||
vllm==0.1.2
|
||||
huggingface-hub==0.16.4
|
||||
runpod @ git+https://github.com/runpod/runpod-python@main#egg=runpod
|
||||
|
||||
+9
-3
@@ -1,5 +1,5 @@
|
||||
#!/usr/bin/env python
|
||||
''' Contains the handler function that will be called by the serverless. '''
|
||||
''' Contains the handler function that will be called by the serverless worker. '''
|
||||
|
||||
# Start the VLLM serving layer on our RunPod worker.
|
||||
from templates import DEFAULT_TEMPLATE, LLAMA_TEMPLATE
|
||||
@@ -9,8 +9,14 @@ import runpod
|
||||
import os
|
||||
|
||||
# Prepare the model and tokenizer
|
||||
MODEL = os.environ['MODEL']
|
||||
TOKENIZER = os.environ['TOKENIZER']
|
||||
MODEL = os.environ.get('MODEL', None)
|
||||
TOKENIZER = os.environ.get('TOKENIZER', None)
|
||||
|
||||
if not MODEL:
|
||||
print("Error: The model has not been provided.")
|
||||
|
||||
if not TOKENIZER:
|
||||
print("Error: The tokenizer has not been provided.")
|
||||
|
||||
# Prepare the engine's arguments
|
||||
engine_args = AsyncEngineArgs(
|
||||
|
||||
Reference in New Issue
Block a user