From e73e108193971da10b2ee5bc931a4196b1d0efc7 Mon Sep 17 00:00:00 2001 From: Jorg Doku Date: Wed, 30 Aug 2023 12:13:20 -0500 Subject: [PATCH] reduce logs --- src/benchmark.py | 5 +++-- src/handler.py | 5 +++-- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/src/benchmark.py b/src/benchmark.py index 40e432b..f83b89b 100644 --- a/src/benchmark.py +++ b/src/benchmark.py @@ -35,11 +35,11 @@ def make_request(url, headers, payload): response = requests.post(url, headers=headers, json=payload) return response -url = "https://api.runpod.ai/v2/4hlrhh430u5tz7/run" +url = "https://api.runpod.ai/v2/4hlrhh430u5tz7/status/1a73bdc5-aede-4c0a-8fa2-2ac2fb3010dc" while True: # Number of concurrent requests to make - num_requests = 100 + num_requests = 1 with concurrent.futures.ThreadPoolExecutor(max_workers=num_requests) as executor: futures = [executor.submit(make_request, url, headers, payload) for _ in range(num_requests)] @@ -48,6 +48,7 @@ while True: for future in concurrent.futures.as_completed(futures): response = future.result() # Handle response as needed + print(response.json()) print(response.status_code) # Sleep for 1 second before starting the next iteration diff --git a/src/handler.py b/src/handler.py index 1364588..a6ed592 100644 --- a/src/handler.py +++ b/src/handler.py @@ -15,6 +15,7 @@ MODEL_NAME = os.environ.get('MODEL_NAME') MODEL_BASE_PATH = os.environ.get('MODEL_BASE_PATH', '/runpod-volume/') STREAMING = os.environ.get('STREAMING', False) == 'True' TOKENIZER = os.environ.get('TOKENIZER', None) +USE_FULL_METRICS = os.environ.get('USE_FULL_METRICS', False) if not MODEL_NAME: print("Error: The model has not been provided.") @@ -173,7 +174,7 @@ async def handler_streaming(job: dict) -> Generator[dict[str, list], None, None] text_outputs.append((" " if text_pos > 0 else "") + text_chunk) # Metrics for the vLLM serverless worker - runpod_metrics = prepare_metrics() + runpod_metrics = prepare_metrics() if USE_FULL_METRICS else {} # The input job runpod_metrics['job_input'] = job_input @@ -257,7 +258,7 @@ async def handler(job: dict) -> dict[str, list]: num_seqs = sampling_params.n # Metrics for the vLLM serverless worker - runpod_metrics = prepare_metrics() + runpod_metrics = prepare_metrics() if USE_FULL_METRICS else {} # The input job runpod_metrics['job_input'] = job_input