Brings forward the L40 GPU type in tests.json and the new llama/qwen tuned config files, while keeping Dockerfile pinned at vllm 0.20.2 (v2.20.1 state). Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
44 lines
978 B
JSON
44 lines
978 B
JSON
{
|
|
"tests": [
|
|
{
|
|
"name": "basic_inference_test",
|
|
"input": {
|
|
"prompt": "Write a short poem about artificial intelligence."
|
|
},
|
|
"timeout": 300000
|
|
},
|
|
{
|
|
"name": "openai_messages_test",
|
|
"input": {
|
|
"openai_route": "/v1/chat/completions",
|
|
"openai_input": {
|
|
"messages": [
|
|
{
|
|
"role": "system",
|
|
"content": "You are a helpful assistant that writes concise responses."
|
|
},
|
|
{
|
|
"role": "user",
|
|
"content": "Explain what a neural network is in one sentence."
|
|
}
|
|
],
|
|
"max_tokens": 200,
|
|
"temperature": 0.1
|
|
}
|
|
},
|
|
"timeout": 300000
|
|
}
|
|
],
|
|
"config": {
|
|
"gpuTypeId": "NVIDIA L40",
|
|
"gpuCount": 1,
|
|
"env": [
|
|
{
|
|
"key": "MODEL_NAME",
|
|
"value": "HuggingFaceTB/SmolLM2-135M-Instruct"
|
|
}
|
|
],
|
|
"allowedCudaVersions": ["13.0"]
|
|
}
|
|
}
|