Final changes
This commit is contained in:
+6
-5
@@ -45,21 +45,22 @@ async def handler(job: dict) -> Generator[str, None, None]:
|
||||
async for request_output in results_generator:
|
||||
for output in request_output.outputs:
|
||||
if streaming:
|
||||
batch.append(output.text[len(last_output_text):])
|
||||
batch.append({"text": output.text[len(last_output_text):]})
|
||||
if len(batch) >= batch_size:
|
||||
yield batch
|
||||
batch = []
|
||||
last_output_text = output.text
|
||||
|
||||
if not streaming:
|
||||
yield [last_output_text]
|
||||
yield [{"text":last_output_text}]
|
||||
|
||||
if batch and streaming:
|
||||
yield batch
|
||||
|
||||
if return_token_counts:
|
||||
yield {"input_tokens": len(output.prompt_token_ids),
|
||||
"output_tokens": len(output.outputs[-1].token_ids)}
|
||||
if return_token_counts and request_output is not None:
|
||||
token_counts = {"token_counts":{"input": len(request_output.prompt_token_ids),
|
||||
"output": len(output.outputs[-1].token_ids)}}
|
||||
yield token_counts
|
||||
|
||||
# Start the serverless worker
|
||||
runpod.serverless.start({
|
||||
|
||||
@@ -1,74 +0,0 @@
|
||||
import time
|
||||
|
||||
# Log every 5 seconds
|
||||
_LOGGING_INTERVAL_SEC = 5
|
||||
|
||||
# https://github.com/vllm-project/vllm/blob/c393af6cd70373ab88b22dbbe59e14bb80ea343d/vllm/engine/llm_engine.py#L342
|
||||
def vllm_log_system_stats(
|
||||
self,
|
||||
prompt_run: bool,
|
||||
num_batched_tokens: int,
|
||||
) -> None:
|
||||
now = time.time()
|
||||
# Log the number of batched input tokens.
|
||||
if prompt_run:
|
||||
self.num_prompt_tokens.append((now, num_batched_tokens))
|
||||
else:
|
||||
self.num_generation_tokens.append((now, num_batched_tokens))
|
||||
|
||||
elapsed_time = now - self.last_logging_time
|
||||
if elapsed_time < _LOGGING_INTERVAL_SEC:
|
||||
return
|
||||
|
||||
# Discard the old stats.
|
||||
self.num_prompt_tokens = [(t, n) for t, n in self.num_prompt_tokens
|
||||
if now - t < _LOGGING_INTERVAL_SEC]
|
||||
self.num_generation_tokens = [(t, n)
|
||||
for t, n in self.num_generation_tokens
|
||||
if now - t < _LOGGING_INTERVAL_SEC]
|
||||
|
||||
if len(self.num_prompt_tokens) > 1:
|
||||
total_num_tokens = sum(n for _, n in self.num_prompt_tokens[:-1])
|
||||
window = now - self.num_prompt_tokens[0][0]
|
||||
avg_prompt_throughput = total_num_tokens / window
|
||||
else:
|
||||
avg_prompt_throughput = 0.0
|
||||
if len(self.num_generation_tokens) > 1:
|
||||
total_num_tokens = sum(n
|
||||
for _, n in self.num_generation_tokens[:-1])
|
||||
window = now - self.num_generation_tokens[0][0]
|
||||
avg_generation_throughput = total_num_tokens / window
|
||||
else:
|
||||
avg_generation_throughput = 0.0
|
||||
|
||||
total_num_gpu_blocks = self.cache_config.num_gpu_blocks
|
||||
num_free_gpu_blocks = (
|
||||
self.scheduler.block_manager.get_num_free_gpu_blocks())
|
||||
num_used_gpu_blocks = total_num_gpu_blocks - num_free_gpu_blocks
|
||||
gpu_cache_usage = num_used_gpu_blocks / total_num_gpu_blocks
|
||||
|
||||
total_num_cpu_blocks = self.cache_config.num_cpu_blocks
|
||||
if total_num_cpu_blocks > 0:
|
||||
num_free_cpu_blocks = (
|
||||
self.scheduler.block_manager.get_num_free_cpu_blocks())
|
||||
num_used_cpu_blocks = total_num_cpu_blocks - num_free_cpu_blocks
|
||||
cpu_cache_usage = num_used_cpu_blocks / total_num_cpu_blocks
|
||||
else:
|
||||
cpu_cache_usage = 0.0
|
||||
|
||||
self.last_logging_time = now
|
||||
|
||||
metrics = {
|
||||
'avg_prompt_throughput': avg_prompt_throughput, # tokens / second
|
||||
'avg_gen_throughput': avg_generation_throughput, # tokens / second
|
||||
'running': len(self.scheduler.running), # number of sequences
|
||||
'swapped': len(self.scheduler.swapped), # number of sequences
|
||||
'pending': len(self.scheduler.waiting), # number of sequences
|
||||
'gpu_kv_cache_usage': gpu_cache_usage, # percentage
|
||||
'cpu_kv_cache_usage': cpu_cache_usage, # percentage
|
||||
}
|
||||
|
||||
# Print metrics
|
||||
print(metrics)
|
||||
|
||||
self.metrics = metrics
|
||||
Reference in New Issue
Block a user