print out llm metrics, make metrics more verbose

This commit is contained in:
Jorg Doku
2023-08-31 19:09:04 -05:00
parent 06183813cf
commit 5e04e5ebb0
2 changed files with 4 additions and 1 deletions
+1 -1
View File
@@ -15,7 +15,7 @@ MODEL_NAME = os.environ.get('MODEL_NAME')
MODEL_BASE_PATH = os.environ.get('MODEL_BASE_PATH', '/runpod-volume/')
STREAMING = os.environ.get('STREAMING', False) == 'True'
TOKENIZER = os.environ.get('TOKENIZER', None)
USE_FULL_METRICS = os.environ.get('USE_FULL_METRICS', False)
USE_FULL_METRICS = os.environ.get('USE_FULL_METRICS', True)
if not MODEL_NAME:
print("Error: The model has not been provided.")
+3
View File
@@ -68,4 +68,7 @@ def vllm_log_system_stats(
'cpu_kv_cache_usage': cpu_cache_usage, # percentage
}
# Print metrics
print(metrics)
self.metrics = metrics