Final changes

This commit is contained in:
alpayariyak
2023-12-08 10:29:24 +00:00
parent 5acbbf5528
commit 8156b1d8e0
2 changed files with 6 additions and 79 deletions
+6 -5
View File
@@ -45,21 +45,22 @@ async def handler(job: dict) -> Generator[str, None, None]:
async for request_output in results_generator:
for output in request_output.outputs:
if streaming:
batch.append(output.text[len(last_output_text):])
batch.append({"text": output.text[len(last_output_text):]})
if len(batch) >= batch_size:
yield batch
batch = []
last_output_text = output.text
if not streaming:
yield [last_output_text]
yield [{"text":last_output_text}]
if batch and streaming:
yield batch
if return_token_counts:
yield {"input_tokens": len(output.prompt_token_ids),
"output_tokens": len(output.outputs[-1].token_ids)}
if return_token_counts and request_output is not None:
token_counts = {"token_counts":{"input": len(request_output.prompt_token_ids),
"output": len(output.outputs[-1].token_ids)}}
yield token_counts
# Start the serverless worker
runpod.serverless.start({
-74
View File
@@ -1,74 +0,0 @@
import time
# Log every 5 seconds
_LOGGING_INTERVAL_SEC = 5
# https://github.com/vllm-project/vllm/blob/c393af6cd70373ab88b22dbbe59e14bb80ea343d/vllm/engine/llm_engine.py#L342
def vllm_log_system_stats(
self,
prompt_run: bool,
num_batched_tokens: int,
) -> None:
now = time.time()
# Log the number of batched input tokens.
if prompt_run:
self.num_prompt_tokens.append((now, num_batched_tokens))
else:
self.num_generation_tokens.append((now, num_batched_tokens))
elapsed_time = now - self.last_logging_time
if elapsed_time < _LOGGING_INTERVAL_SEC:
return
# Discard the old stats.
self.num_prompt_tokens = [(t, n) for t, n in self.num_prompt_tokens
if now - t < _LOGGING_INTERVAL_SEC]
self.num_generation_tokens = [(t, n)
for t, n in self.num_generation_tokens
if now - t < _LOGGING_INTERVAL_SEC]
if len(self.num_prompt_tokens) > 1:
total_num_tokens = sum(n for _, n in self.num_prompt_tokens[:-1])
window = now - self.num_prompt_tokens[0][0]
avg_prompt_throughput = total_num_tokens / window
else:
avg_prompt_throughput = 0.0
if len(self.num_generation_tokens) > 1:
total_num_tokens = sum(n
for _, n in self.num_generation_tokens[:-1])
window = now - self.num_generation_tokens[0][0]
avg_generation_throughput = total_num_tokens / window
else:
avg_generation_throughput = 0.0
total_num_gpu_blocks = self.cache_config.num_gpu_blocks
num_free_gpu_blocks = (
self.scheduler.block_manager.get_num_free_gpu_blocks())
num_used_gpu_blocks = total_num_gpu_blocks - num_free_gpu_blocks
gpu_cache_usage = num_used_gpu_blocks / total_num_gpu_blocks
total_num_cpu_blocks = self.cache_config.num_cpu_blocks
if total_num_cpu_blocks > 0:
num_free_cpu_blocks = (
self.scheduler.block_manager.get_num_free_cpu_blocks())
num_used_cpu_blocks = total_num_cpu_blocks - num_free_cpu_blocks
cpu_cache_usage = num_used_cpu_blocks / total_num_cpu_blocks
else:
cpu_cache_usage = 0.0
self.last_logging_time = now
metrics = {
'avg_prompt_throughput': avg_prompt_throughput, # tokens / second
'avg_gen_throughput': avg_generation_throughput, # tokens / second
'running': len(self.scheduler.running), # number of sequences
'swapped': len(self.scheduler.swapped), # number of sequences
'pending': len(self.scheduler.waiting), # number of sequences
'gpu_kv_cache_usage': gpu_cache_usage, # percentage
'cpu_kv_cache_usage': cpu_cache_usage, # percentage
}
# Print metrics
print(metrics)
self.metrics = metrics