Files
worker-vllm/.runpod/tests.json
T

44 lines
979 B
JSON

{
"tests": [
{
"name": "basic_inference_test",
"input": {
"prompt": "Write a short poem about artificial intelligence."
},
"timeout": 300000
},
{
"name": "openai_messages_test",
"input": {
"openai_route": "/v1/chat/completions",
"openai_input": {
"messages": [
{
"role": "system",
"content": "You are a helpful assistant that writes concise responses."
},
{
"role": "user",
"content": "Explain what a neural network is in one sentence."
}
],
"max_tokens": 200,
"temperature": 0.1
}
},
"timeout": 300000
}
],
"config": {
"gpuTypeId": "NNVIDIA L40",
"gpuCount": 1,
"env": [
{
"key": "MODEL_NAME",
"value": "HuggingFaceTB/SmolLM2-135M-Instruct"
}
],
"allowedCudaVersions": ["13.0"]
}
}