reduce logs

This commit is contained in:
Jorg Doku
2023-08-30 12:13:20 -05:00
parent be4421e8e2
commit e73e108193
2 changed files with 6 additions and 4 deletions
+3 -2
View File
@@ -35,11 +35,11 @@ def make_request(url, headers, payload):
response = requests.post(url, headers=headers, json=payload)
return response
url = "https://api.runpod.ai/v2/4hlrhh430u5tz7/run"
url = "https://api.runpod.ai/v2/4hlrhh430u5tz7/status/1a73bdc5-aede-4c0a-8fa2-2ac2fb3010dc"
while True:
# Number of concurrent requests to make
num_requests = 100
num_requests = 1
with concurrent.futures.ThreadPoolExecutor(max_workers=num_requests) as executor:
futures = [executor.submit(make_request, url, headers, payload) for _ in range(num_requests)]
@@ -48,6 +48,7 @@ while True:
for future in concurrent.futures.as_completed(futures):
response = future.result()
# Handle response as needed
print(response.json())
print(response.status_code)
# Sleep for 1 second before starting the next iteration
+3 -2
View File
@@ -15,6 +15,7 @@ MODEL_NAME = os.environ.get('MODEL_NAME')
MODEL_BASE_PATH = os.environ.get('MODEL_BASE_PATH', '/runpod-volume/')
STREAMING = os.environ.get('STREAMING', False) == 'True'
TOKENIZER = os.environ.get('TOKENIZER', None)
USE_FULL_METRICS = os.environ.get('USE_FULL_METRICS', False)
if not MODEL_NAME:
print("Error: The model has not been provided.")
@@ -173,7 +174,7 @@ async def handler_streaming(job: dict) -> Generator[dict[str, list], None, None]
text_outputs.append((" " if text_pos > 0 else "") + text_chunk)
# Metrics for the vLLM serverless worker
runpod_metrics = prepare_metrics()
runpod_metrics = prepare_metrics() if USE_FULL_METRICS else {}
# The input job
runpod_metrics['job_input'] = job_input
@@ -257,7 +258,7 @@ async def handler(job: dict) -> dict[str, list]:
num_seqs = sampling_params.n
# Metrics for the vLLM serverless worker
runpod_metrics = prepare_metrics()
runpod_metrics = prepare_metrics() if USE_FULL_METRICS else {}
# The input job
runpod_metrics['job_input'] = job_input