Compare commits

..
17 Commits
Author SHA1 Message Date
chrisvelaandGitHub 76f345be08 Merge pull request #321 from runpod-workers/revert-318-feat/0.24.0
Release / release (push) Waiting to run
Revert "feat: update to 0.24.0"
2026-07-24 15:23:50 -05:00
chrisvelaandGitHub 4dfda80fd0 Revert "feat: update to 0.24.0" 2026-07-24 15:16:55 -05:00
Dean QuiñanolaandGitHub 89f25b0d70 Merge pull request #319 from runpod-workers/runpod-package-update
chore: update runpod to 1.11.0
2026-07-24 12:13:02 -07:00
deanqandgithub-actions[bot] 4808160735 chore: update runpod to 1.11.0 2026-07-24 18:30:38 +00:00
Dean QuiñanolaandGitHub 51e1aee770 Merge pull request #318 from runpod-workers/feat/0.24.0
feat: update to 0.24.0
2026-07-24 11:30:29 -07:00
velaraptor-runpod 86b7951c39 feat: update to 0.24.0 2026-07-16 18:33:38 -05:00
chrisvelaandGitHub e786204cf5 Merge pull request #316 from runpod-workers/docs/sync-vllm-version-0.23.0
docs: sync vLLM version to 0.23.0 in READMEs
2026-07-10 11:10:53 -05:00
chrisvelaandGitHub 591f9d6531 Merge pull request #317 from runpod-workers/fix/test-sls-gh-action
fix: fix image to test with sls in gh actions
2026-07-10 09:55:53 -05:00
velaraptor-runpod 49967d0c09 fix: fix image to test with sls in gh actions 2026-07-09 22:25:40 -05:00
velaraptor-runpodandgithub-actions[bot] e97cab854f docs: sync vLLM version to 0.23.0 in READMEs 2026-07-10 02:18:51 +00:00
chrisvelaandGitHub 828e797498 Merge pull request #315 from runpod-workers/feat/0.23.0
feat: update vllm to 0.23.0 + add sls e2e model testing
2026-07-09 21:18:41 -05:00
velaraptor-runpod 373847d9f9 fix: pin models to version, except for gpt-oss 2026-07-09 19:21:24 -05:00
velaraptor-runpod 2000acc0a3 chore: add timeout for e2e serverless 2026-07-09 19:15:19 -05:00
velaraptor-runpod 89e3b2e920 fix: fix for huggingface, secret keys 2026-07-09 18:56:58 -05:00
velaraptor-runpod 85f8c81c1c add sls e2e tests 2026-07-09 18:22:57 -05:00
velaraptor-runpod 88490e9551 feat: update to 0.23.0 2026-07-09 18:22:40 -05:00
velaraptor-runpod 798f2b12eb feat: update to 0.23.0 2026-07-01 16:19:01 -05:00
11 changed files with 525 additions and 11 deletions
@@ -0,0 +1,162 @@
name: Serverless Model Tests
on:
pull_request:
branches:
- "**"
push:
branches:
- "main"
permissions:
contents: read
# Only one run per ref at a time — these tests spin up real H100 endpoints.
concurrency:
group: serverless-model-tests-${{ github.ref }}
cancel-in-progress: true
jobs:
# On PRs we only smoke-test one (small, cheap) model to keep review loops fast.
test-pr:
if: github.event_name == 'pull_request'
runs-on: [blacksmith-8vcpu-ubuntu-2204, linux]
timeout-minutes: 60
steps:
- name: Checkout
uses: actions/checkout@v4
- name: Set up QEMU
uses: docker/setup-qemu-action@v3
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3
- name: Login to Docker Hub
uses: docker/login-action@v3
with:
username: ${{ secrets.DOCKERHUB_USERNAME }}
password: ${{ secrets.DOCKERHUB_TOKEN }}
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: "3.11"
- name: Install script dependencies
run: pip install pyyaml
- name: Run serverless e2e test
id: e2e_test
run: python scripts/serverless_e2e_test.py --config configs/gpt-oss/gpt_oss_120b.yaml --build
env:
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY_2 }}
DOCKERHUB_REPO: ${{ vars.DOCKERHUB_REPO || 'runpod' }}
DOCKERHUB_IMG: ${{ vars.DOCKERHUB_IMG || 'worker-v1-vllm' }}
HUGGINGFACE_ACCESS_TOKEN: ${{ secrets.HUGGINGFACE_ACCESS_TOKEN_2 }}
- name: Cleanup safety net
if: always()
run: |
if [ -n "${{ steps.e2e_test.outputs.endpoint_id }}" ]; then
curl -sf -X DELETE "https://rest.runpod.io/v1/endpoints/${{ steps.e2e_test.outputs.endpoint_id }}" \
-H "Authorization: Bearer ${{ secrets.RUNPOD_API_KEY_2 }}" || true
fi
if [ -n "${{ steps.e2e_test.outputs.template_id }}" ]; then
curl -sf -X DELETE "https://rest.runpod.io/v1/templates/${{ steps.e2e_test.outputs.template_id }}" \
-H "Authorization: Bearer ${{ secrets.RUNPOD_API_KEY_2 }}" || true
fi
# On push to main we test every tuned model config in parallel.
discover:
if: github.event_name == 'push'
runs-on: ubuntu-latest
outputs:
configs: ${{ steps.list.outputs.configs }}
steps:
- name: Checkout
uses: actions/checkout@v4
- name: List model configs
id: list
run: |
CONFIGS=$(find configs -name '*.yaml' | sort | jq -R -s -c 'split("\n") | map(select(length > 0))')
echo "configs=$CONFIGS" >> "$GITHUB_OUTPUT"
# The image is the same regardless of which model config is under test (configs
# only supply endpoint env vars), so build/push it once and let every matrix job
# in test-main reuse that same tag instead of rebuilding per config.
build:
if: github.event_name == 'push'
runs-on: [blacksmith-8vcpu-ubuntu-2204, linux]
outputs:
release_version: ${{ steps.build.outputs.release_version }}
steps:
- name: Checkout
uses: actions/checkout@v4
- name: Set up QEMU
uses: docker/setup-qemu-action@v3
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3
- name: Login to Docker Hub
uses: docker/login-action@v3
with:
username: ${{ secrets.DOCKERHUB_USERNAME }}
password: ${{ secrets.DOCKERHUB_TOKEN }}
- name: Build and push image
id: build
env:
DOCKERHUB_REPO: ${{ vars.DOCKERHUB_REPO || 'runpod' }}
DOCKERHUB_IMG: ${{ vars.DOCKERHUB_IMG || 'worker-v1-vllm' }}
RELEASE_VERSION: test-${{ github.sha }}
HUGGINGFACE_ACCESS_TOKEN: ${{ secrets.HUGGINGFACE_ACCESS_TOKEN }}
run: |
docker buildx bake --push
echo "release_version=${RELEASE_VERSION}" >> "$GITHUB_OUTPUT"
test-main:
needs: [discover, build]
if: github.event_name == 'push' && needs.discover.result == 'success' && needs.build.result == 'success'
runs-on: [blacksmith-8vcpu-ubuntu-2204, linux]
timeout-minutes: 60
strategy:
fail-fast: false
matrix:
config: ${{ fromJSON(needs.discover.outputs.configs) }}
steps:
- name: Checkout
uses: actions/checkout@v4
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: "3.11"
- name: Install script dependencies
run: pip install pyyaml
- name: Run serverless e2e test
id: e2e_test
run: python scripts/serverless_e2e_test.py --config "${{ matrix.config }}" --image "${DOCKERHUB_REPO}/${DOCKERHUB_IMG}:${RELEASE_VERSION}"
env:
RUNPOD_API_KEY: ${{ secrets.RUNPOD_API_KEY_2 }}
HUGGINGFACE_ACCESS_TOKEN: ${{ secrets.HUGGINGFACE_ACCESS_TOKEN_2 }}
DOCKERHUB_REPO: ${{ vars.DOCKERHUB_REPO || 'runpod' }}
DOCKERHUB_IMG: ${{ vars.DOCKERHUB_IMG || 'worker-v1-vllm' }}
RELEASE_VERSION: ${{ needs.build.outputs.release_version }}
- name: Cleanup safety net
if: always()
run: |
if [ -n "${{ steps.e2e_test.outputs.endpoint_id }}" ]; then
curl -sf -X DELETE "https://rest.runpod.io/v1/endpoints/${{ steps.e2e_test.outputs.endpoint_id }}" \
-H "Authorization: Bearer ${{ secrets.RUNPOD_API_KEY_2 }}" || true
fi
if [ -n "${{ steps.e2e_test.outputs.template_id }}" ]; then
curl -sf -X DELETE "https://rest.runpod.io/v1/templates/${{ steps.e2e_test.outputs.template_id }}" \
-H "Authorization: Bearer ${{ secrets.RUNPOD_API_KEY_2 }}" || true
fi
+1 -1
View File
@@ -6,7 +6,7 @@ Run LLMs using [vLLM](https://docs.vllm.ai) with an OpenAI-compatible API
[![RunPod](https://api.runpod.io/badge/runpod-workers/worker-vllm)](https://www.runpod.io/console/hub/runpod-workers/worker-vllm)
Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
Current vLLM version: [0.23.0](https://github.com/vllm-project/vllm/releases/tag/v0.23.0)
---
+1
View File
@@ -247,6 +247,7 @@
"name": "Max Parallel Loading Workers",
"type": "number",
"description": "Load model sequentially in multiple batches.",
"default": 1,
"advanced": true
}
},
+26 -6
View File
@@ -1,17 +1,36 @@
FROM nvidia/cuda:13.0.2-devel-ubuntu22.04
ENV DEBIAN_FRONTEND=noninteractive
RUN apt-get update -y \
&& apt-get install -y python3-pip curl git \
&& curl -LsSf https://astral.sh/uv/install.sh | sh
&& apt-get install -y curl git software-properties-common \
&& add-apt-repository -y ppa:deadsnakes/ppa \
&& apt-get install -y python3.12 python3.12-dev python3.12-venv \
&& update-alternatives --install /usr/bin/python3 python3 /usr/bin/python3.12 1 \
&& update-alternatives --set python3 /usr/bin/python3.12 \
&& rm -f /usr/lib/python3.12/EXTERNALLY-MANAGED \
&& curl -LsSf https://astral.sh/uv/install.sh | sh
ENV PATH="/root/.local/bin:$PATH"
RUN ldconfig /usr/local/cuda-13.0/compat/
# Install vLLM with FlashInfer - use CUDA 130 PyTorch wheels
RUN uv pip install --system "packaging>=24.2" && \
uv pip install --system "vllm[flashinfer]==0.20.2" && \
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation
RUN --mount=type=cache,target=/root/.cache/uv \
uv pip install --system "packaging>=24.2" && \
uv pip install --system "vllm[flashinfer]==0.23.0" && \
uv pip install --system git+https://github.com/deepseek-ai/DeepGEMM.git@714dd1a4a980f7937a74343d19a8eba4fe321480 --no-build-isolation && \
uv pip install --system --force-reinstall --no-deps nixl-cu13
# Fix CUTLASS DSL cu13 install order: nvidia-cutlass-dsl[cu13] installs
# -libs-base and -libs-cu13 wheels that share paths with different content.
# uv can extract them in either order, leaving base files that break CUDA 13
# CuTe DSL JIT. Force -libs-cu13 last. See vllm-project/vllm#45204.
RUN CUTLASS_DSL_VERSION=$(uv pip show --system nvidia-cutlass-dsl 2>/dev/null | awk '/^Version:/{print $2}') && \
if [ -n "$CUTLASS_DSL_VERSION" ]; then \
uv pip install --system --force-reinstall --no-deps \
"nvidia-cutlass-dsl-libs-cu13==${CUTLASS_DSL_VERSION}"; \
fi
# Install additional Python dependencies (after vLLM to avoid PyTorch version conflicts)
COPY builder/requirements.txt /requirements.txt
@@ -47,7 +66,8 @@ ENV MODEL_NAME=$MODEL_NAME \
# Disable DeepGEMM MoE kernels by default; override with VLLM_USE_DEEP_GEMM=1 to enable
VLLM_USE_DEEP_GEMM=0
ENV PYTHONPATH="/:/vllm-workspace"
ENV PYTHONPATH="/:/vllm-workspace" \
LD_LIBRARY_PATH="/usr/local/nvidia/lib64:/usr/local/cuda/lib64:${LD_LIBRARY_PATH}"
RUN if [ "${VLLM_NIGHTLY}" = "true" ]; then \
uv pip install --system -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \
+1 -1
View File
@@ -8,7 +8,7 @@ Deploy OpenAI-Compatible Blazing-Fast LLM Endpoints powered by the [vLLM](https:
![vLLM worker banner](https://image.runpod.ai/preview/vllm/vllm-banner.png)
Current vLLM version: [0.20.2](https://github.com/vllm-project/vllm/releases/tag/v0.20.2)
Current vLLM version: [0.23.0](https://github.com/vllm-project/vllm/releases/tag/v0.23.0)
> Check out our Load Balancer implementation here: [vLLM Load Balancer](https://github.com/runpod-workers/vllm-loadbalancer-ep)
+2 -2
View File
@@ -1,9 +1,9 @@
ray
pandas
pyarrow
runpod~=1.10.0
runpod~=1.11.0
huggingface-hub
lmcache==0.4.5
lmcache==0.5.0
packaging>=24.2
typing-extensions>=4.8.0
pydantic
+1
View File
@@ -8,4 +8,5 @@ kv-cache-dtype: fp8
enforce-eager: false
enable-prefix-caching: true
enable-chunked-prefill: true
vllm-release: v2.22.5
speculative-config: '{"model":"RedHatAI/gemma-4-31B-it-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
+1
View File
@@ -6,5 +6,6 @@ trust-remote-code: true
quantization: fp8
kv-cache-dtype: fp8
enforce-eager: false
vllm-release: v2.22.5
enable-prefix-caching: true
speculative-config: '{"model":"RedHatAI/Llama-3.1-8B-Instruct-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
+2
View File
@@ -7,4 +7,6 @@ quantization: fp8
kv-cache-dtype: fp8
enforce-eager: false
enable-prefix-caching: true
vllm-release: v2.22.5
compilation-config: '{"cudagraph_mode": "PIECEWISE"}'
speculative-config: '{"model":"RedHatAI/Qwen3-8B-speculator.eagle3","method":"eagle3","num_speculative_tokens":3}'
+328
View File
@@ -0,0 +1,328 @@
#!/usr/bin/env python3
"""End-to-end test: build/push the worker image, deploy it as a real RunPod
Serverless endpoint on H100, run the .runpod/tests.json test cases against it,
then tear the endpoint down.
Usage:
RUNPOD_API_KEY=... python scripts/serverless_e2e_test.py \\
--config configs/qwen/qwen3_8b.yaml --build
RUNPOD_API_KEY=... python scripts/serverless_e2e_test.py \\
--config configs/qwen/qwen3_8b.yaml --image runpod/worker-v1-vllm:dev-my-branch
"""
import argparse
import json
import os
import subprocess
import sys
import time
import urllib.error
import urllib.request
import uuid
from pathlib import Path
import yaml
REPO_ROOT = Path(__file__).resolve().parent.parent
REST_API_BASE = "https://rest.runpod.io/v1"
JOB_API_BASE = "https://api.runpod.ai/v2"
DEFAULT_GPU_TYPE_IDS = [
"NVIDIA H100 80GB HBM3",
"NVIDIA H100 NVL",
"NVIDIA H100 PCIe",
]
TERMINAL_STATUSES = {"COMPLETED", "FAILED", "TIMED_OUT", "CANCELLED"}
def log(msg: str) -> None:
print(f"[serverless_e2e_test] {msg}", flush=True)
def write_github_output(key: str, value: str) -> None:
"""Expose a value to later CI steps (e.g. an always() cleanup safety net
for when this process gets killed before its own `finally` can run)."""
path = os.environ.get("GITHUB_OUTPUT")
if not path:
return
with open(path, "a") as f:
f.write(f"{key}={value}\n")
def stringify_env_value(value) -> str:
if isinstance(value, bool):
return "true" if value else "false"
if isinstance(value, (dict, list)):
return json.dumps(value)
return str(value)
def load_hub_defaults(hub_json_path: Path) -> dict:
hub = json.loads(hub_json_path.read_text())
config = hub.get("config", {})
base_env = {}
for entry in config.get("env", []):
default = entry.get("input", {}).get("default")
if default is None:
continue
base_env[entry["key"]] = stringify_env_value(default)
return {
"containerDiskInGb": config.get("containerDiskInGb", 50),
"gpuCount": config.get("gpuCount", 1),
"allowedCudaVersions": config.get("allowedCudaVersions"),
"minCudaVersion": config.get("minCudaVersion"),
"env": base_env,
}
# config keys whose naive `KEY.replace("-", "_").upper()` transform doesn't match
# the env var the worker actually reads (see src/engine_args.py ENV_ALIASES).
CONFIG_KEY_ALIASES = {
"MODEL": "MODEL_NAME",
}
def load_model_env(config_yaml_path: Path) -> dict:
raw = yaml.safe_load(config_yaml_path.read_text()) or {}
env = {}
for key, value in raw.items():
env_key = key.replace("-", "_").upper()
env_key = CONFIG_KEY_ALIASES.get(env_key, env_key)
env[env_key] = stringify_env_value(value)
return env
def build_and_push_image(dockerhub_repo: str, dockerhub_img: str, release_version: str) -> str:
hf_token = os.environ.get("HUGGINGFACE_ACCESS_TOKEN", "")
dockerhub_user = os.environ.get("DOCKERHUB_USERNAME")
dockerhub_pass = os.environ.get("DOCKERHUB_TOKEN")
if dockerhub_user and dockerhub_pass:
log(f"Logging in to Docker Hub as {dockerhub_user}")
subprocess.run(
["docker", "login", "-u", dockerhub_user, "--password-stdin"],
input=dockerhub_pass,
text=True,
check=True,
cwd=REPO_ROOT,
)
else:
log("DOCKERHUB_USERNAME/DOCKERHUB_TOKEN not set; assuming docker is already logged in")
tag = f"{dockerhub_repo}/{dockerhub_img}:{release_version}"
log(f"Building and pushing {tag} via docker buildx bake")
# docker-bake.hcl's DOCKERHUB_REPO/DOCKERHUB_IMG/RELEASE_VERSION are bake-level
# `variable` blocks that compute `tags` - they read from env vars of the same
# name, NOT from `--set target.args.*` (that sets Dockerfile build ARGs, a
# separate namespace the Dockerfile doesn't even declare these under).
bake_env = {
**os.environ,
"DOCKERHUB_REPO": dockerhub_repo,
"DOCKERHUB_IMG": dockerhub_img,
"RELEASE_VERSION": release_version,
"HUGGINGFACE_ACCESS_TOKEN": hf_token,
}
subprocess.run(
["docker", "buildx", "bake", "--push"],
check=True,
cwd=REPO_ROOT,
env=bake_env,
)
return tag
def api_request(method: str, url: str, api_key: str, body: dict | None = None) -> dict:
data = json.dumps(body).encode() if body is not None else None
req = urllib.request.Request(url, data=data, method=method)
req.add_header("Authorization", f"Bearer {api_key}")
req.add_header("Content-Type", "application/json")
try:
with urllib.request.urlopen(req) as resp:
raw = resp.read()
return json.loads(raw) if raw else {}
except urllib.error.HTTPError as e:
detail = e.read().decode(errors="replace")
raise RuntimeError(f"{method} {url} -> HTTP {e.code}: {detail}") from e
def create_template(api_key: str, name: str, image: str, env: dict, container_disk_in_gb: int) -> str:
log(f"Creating template {name!r} for image {image}")
resp = api_request("POST", f"{REST_API_BASE}/templates", api_key, {
"name": name,
"imageName": image,
"isServerless": True,
"env": env,
"containerDiskInGb": container_disk_in_gb,
})
return resp["id"]
def create_endpoint(
api_key: str,
name: str,
template_id: str,
gpu_type_ids: list[str],
gpu_count: int,
allowed_cuda_versions: list[str] | None,
min_cuda_version: str | None,
idle_timeout: int,
) -> str:
log(f"Creating endpoint {name!r} (gpuTypeIds={gpu_type_ids})")
body = {
"name": name,
"templateId": template_id,
"gpuTypeIds": gpu_type_ids,
"gpuCount": gpu_count,
"workersMin": 0,
"workersMax": 1,
"idleTimeout": idle_timeout,
"scalerType": "QUEUE_DELAY",
"scalerValue": 4,
}
if allowed_cuda_versions:
body["allowedCudaVersions"] = allowed_cuda_versions
if min_cuda_version:
body["minCudaVersion"] = min_cuda_version
resp = api_request("POST", f"{REST_API_BASE}/endpoints", api_key, body)
return resp["id"]
def delete_endpoint(api_key: str, endpoint_id: str) -> None:
log(f"Deleting endpoint {endpoint_id}")
try:
api_request("DELETE", f"{REST_API_BASE}/endpoints/{endpoint_id}", api_key)
except RuntimeError as e:
log(f"WARNING: failed to delete endpoint {endpoint_id}: {e}")
def delete_template(api_key: str, template_id: str) -> None:
log(f"Deleting template {template_id}")
try:
api_request("DELETE", f"{REST_API_BASE}/templates/{template_id}", api_key)
except RuntimeError as e:
log(f"WARNING: failed to delete template {template_id}: {e}")
def response_has_error(output) -> bool:
if isinstance(output, dict):
return "error" in output
if isinstance(output, list):
return any(isinstance(item, dict) and "error" in item for item in output)
return False
def run_test_case(api_key: str, endpoint_id: str, test: dict, cold_start_buffer_seconds: int) -> bool:
name = test.get("name", "unnamed_test")
deadline = time.monotonic() + test.get("timeout", 300000) / 1000 + cold_start_buffer_seconds
log(f"Submitting job for test {name!r}")
submit = api_request("POST", f"{JOB_API_BASE}/{endpoint_id}/run", api_key, {"input": test["input"]})
job_id = submit["id"]
while True:
if time.monotonic() > deadline:
log(f"FAIL {name}: timed out waiting for job {job_id}")
return False
status_resp = api_request("GET", f"{JOB_API_BASE}/{endpoint_id}/status/{job_id}", api_key)
status = status_resp.get("status")
if status in TERMINAL_STATUSES:
if status != "COMPLETED":
log(f"FAIL {name}: job {job_id} ended with status {status}: {status_resp}")
return False
if response_has_error(status_resp.get("output")):
log(f"FAIL {name}: job {job_id} completed but output contained an error: {status_resp.get('output')}")
return False
log(f"PASS {name}")
return True
time.sleep(5)
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--config", required=True, type=Path, help="Path to a configs/**/*.yaml vLLM config")
parser.add_argument("--image", help="Existing image tag to test; skips building unless --build is also given")
parser.add_argument("--build", action="store_true", help="Build and push the image before testing")
parser.add_argument("--keep", action="store_true", help="Don't tear down the endpoint/template afterward")
parser.add_argument("--hub-json", type=Path, default=REPO_ROOT / ".runpod" / "hub.json")
parser.add_argument("--tests-file", type=Path, default=REPO_ROOT / ".runpod" / "tests.json")
parser.add_argument("--gpu-type-ids", default=",".join(DEFAULT_GPU_TYPE_IDS))
parser.add_argument("--min-cuda-version", help="Overrides the config/hub.json minCudaVersion")
parser.add_argument("--dockerhub-repo", default=os.environ.get("DOCKERHUB_REPO", "runpod"))
parser.add_argument("--dockerhub-img", default=os.environ.get("DOCKERHUB_IMG", "worker-v1-vllm"))
parser.add_argument("--idle-timeout", type=int, default=60)
parser.add_argument("--cold-start-buffer-seconds", type=int, default=600)
args = parser.parse_args()
api_key = os.environ.get("RUNPOD_API_KEY")
if not api_key:
log("ERROR: RUNPOD_API_KEY is not set")
return 1
model_slug = args.config.stem
run_id = uuid.uuid4().hex[:8]
if args.build or not args.image:
release_version = f"test-{model_slug}-{run_id}"
image = build_and_push_image(args.dockerhub_repo, args.dockerhub_img, release_version)
else:
image = args.image
hub_defaults = load_hub_defaults(args.hub_json)
model_env = load_model_env(args.config)
env = {**hub_defaults["env"], **model_env}
hf_token = os.environ.get("HF_TOKEN") or os.environ.get("HUGGINGFACE_ACCESS_TOKEN")
if hf_token:
env["HF_TOKEN"] = hf_token
tests_data = json.loads(args.tests_file.read_text())
gpu_type_ids = [g.strip() for g in args.gpu_type_ids.split(",") if g.strip()]
resource_name = f"worker-vllm-e2e-{model_slug}-{run_id}"
template_id = None
endpoint_id = None
try:
template_id = create_template(
api_key, resource_name, image, env, hub_defaults["containerDiskInGb"]
)
write_github_output("template_id", template_id)
endpoint_id = create_endpoint(
api_key,
resource_name,
template_id,
gpu_type_ids,
hub_defaults["gpuCount"],
hub_defaults["allowedCudaVersions"],
args.min_cuda_version or hub_defaults["minCudaVersion"],
args.idle_timeout,
)
write_github_output("endpoint_id", endpoint_id)
results = [
run_test_case(api_key, endpoint_id, test, args.cold_start_buffer_seconds)
for test in tests_data["tests"]
]
if all(results):
log(f"All {len(results)} test(s) passed for {model_slug}")
return 0
log(f"{results.count(False)}/{len(results)} test(s) failed for {model_slug}")
return 1
finally:
if not args.keep:
if endpoint_id:
delete_endpoint(api_key, endpoint_id)
if template_id:
delete_template(api_key, template_id)
else:
log(f"--keep passed; leaving endpoint={endpoint_id} template={template_id} running")
if __name__ == "__main__":
sys.exit(main())
-1
View File
@@ -9,7 +9,6 @@ from typing import AsyncGenerator, Optional
from dotenv import load_dotenv
from vllm import AsyncLLMEngine
from vllm.inputs import TextPrompt
from vllm.entrypoints.logger import RequestLogger
from vllm.entrypoints.anthropic.protocol import AnthropicMessagesRequest, AnthropicMessagesResponse, AnthropicError, AnthropicErrorResponse
from vllm.entrypoints.anthropic.serving import AnthropicServingMessages
from vllm.entrypoints.openai.chat_completion.protocol import ChatCompletionRequest