feat: prepare worker-vllm for the hub (#214)
Release / release (push) Waiting to run
Release / release (push) Waiting to run
* docs: remove outdated video; remove old info; added missing config for tools * ci: use proper release for dev (pr only) and production (release only) * ci(hub): added openai example; use smollm2 as base model * docs: added conventions to be able to work with ai ide's * chore: remove outdated stuff * chore: update copyright to 2025 * ci: added github permissions * feat: added gpuIds, gputCount and allowedCudaVersions; removed default value for LOAD_FORMAT to check which influence this has on the ui --------- Co-authored-by: Tim Pietrusky <tim.pietrusky@runpod.io>
This commit is contained in:
co-authored by
Tim Pietrusky
parent
aef1187a30
commit
5f0fc69d75
+13
-1
@@ -7,6 +7,19 @@
|
||||
"config": {
|
||||
"runsOn": "GPU",
|
||||
"containerDiskInGb": 200,
|
||||
"gpuIds": "ADA_80_PRO, AMPERE_80",
|
||||
"gpuCount": 1,
|
||||
"allowedCudaVersions": [
|
||||
"12.9",
|
||||
"12.8",
|
||||
"12.7",
|
||||
"12.6",
|
||||
"12.5",
|
||||
"12.4",
|
||||
"12.3",
|
||||
"12.2",
|
||||
"12.1"
|
||||
],
|
||||
"presets": [
|
||||
{
|
||||
"name": "deepseek-ai/deepseek-r1-distill-llama-8b",
|
||||
@@ -129,7 +142,6 @@
|
||||
"value": "bitsandbytes"
|
||||
}
|
||||
],
|
||||
"default": "auto",
|
||||
"advanced": true
|
||||
}
|
||||
},
|
||||
|
||||
+47
-27
@@ -1,32 +1,52 @@
|
||||
{
|
||||
"tests": [
|
||||
{
|
||||
"name": "basic_inference_test",
|
||||
"input": {
|
||||
"prompt": "Write a short poem about artificial intelligence."
|
||||
},
|
||||
"timeout": 30000
|
||||
}
|
||||
],
|
||||
"config": {
|
||||
"gpuTypeId": "NVIDIA GeForce RTX 4090",
|
||||
"gpuCount": 1,
|
||||
"env": [
|
||||
"tests": [
|
||||
{
|
||||
"name": "basic_inference_test",
|
||||
"input": {
|
||||
"prompt": "Write a short poem about artificial intelligence."
|
||||
},
|
||||
"timeout": 30000
|
||||
},
|
||||
{
|
||||
"name": "openai_messages_test",
|
||||
"input": {
|
||||
"openai_route": "/v1/chat/completions",
|
||||
"openai_input": {
|
||||
"model": "HuggingFaceTB/SmolLM2-135M-Instruct",
|
||||
"messages": [
|
||||
{
|
||||
"key": "MODEL_NAME",
|
||||
"value": "facebook/opt-350m"
|
||||
"role": "system",
|
||||
"content": "You are a helpful assistant that writes concise responses."
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "Explain what a neural network is in one sentence."
|
||||
}
|
||||
],
|
||||
"allowedCudaVersions": [
|
||||
"12.7",
|
||||
"12.6",
|
||||
"12.5",
|
||||
"12.4",
|
||||
"12.3",
|
||||
"12.2",
|
||||
"12.1",
|
||||
"12.0",
|
||||
"11.7"
|
||||
]
|
||||
],
|
||||
"max_tokens": 50,
|
||||
"temperature": 0.7
|
||||
}
|
||||
},
|
||||
"timeout": 30000
|
||||
}
|
||||
],
|
||||
"config": {
|
||||
"gpuTypeId": "NVIDIA GeForce RTX 4090",
|
||||
"gpuCount": 1,
|
||||
"env": [
|
||||
{
|
||||
"key": "MODEL_NAME",
|
||||
"value": "HuggingFaceTB/SmolLM2-135M-Instruct"
|
||||
}
|
||||
],
|
||||
"allowedCudaVersions": [
|
||||
"12.7",
|
||||
"12.6",
|
||||
"12.5",
|
||||
"12.4",
|
||||
"12.3",
|
||||
"12.2",
|
||||
"12.1"
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user