[Bench] Add consistent p90/p95/p99 percentiles for all latency metrics (#27662)
This commit is contained in:
@@ -889,37 +889,56 @@ ASYNC_REQUEST_FUNCS = {
|
|||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class BenchmarkMetrics:
|
class BenchmarkMetrics:
|
||||||
|
# Request counts and token totals
|
||||||
completed: int
|
completed: int
|
||||||
total_input: int
|
total_input: int
|
||||||
total_input_text: int
|
total_input_text: int
|
||||||
total_input_vision: int
|
total_input_vision: int
|
||||||
total_output: int
|
total_output: int
|
||||||
total_output_retokenized: int
|
total_output_retokenized: int
|
||||||
|
|
||||||
|
# Throughput (req/s and tok/s)
|
||||||
request_throughput: float
|
request_throughput: float
|
||||||
input_throughput: float
|
input_throughput: float
|
||||||
output_throughput: float
|
output_throughput: float
|
||||||
output_throughput_retokenized: float
|
output_throughput_retokenized: float
|
||||||
total_throughput: float
|
total_throughput: float
|
||||||
total_throughput_retokenized: float
|
total_throughput_retokenized: float
|
||||||
|
|
||||||
|
# TTFT - Time to First Token (ms)
|
||||||
mean_ttft_ms: float
|
mean_ttft_ms: float
|
||||||
median_ttft_ms: float
|
median_ttft_ms: float
|
||||||
std_ttft_ms: float
|
std_ttft_ms: float
|
||||||
|
p90_ttft_ms: float
|
||||||
|
p95_ttft_ms: float
|
||||||
p99_ttft_ms: float
|
p99_ttft_ms: float
|
||||||
|
|
||||||
|
# TPOT - Time per Output Token, excluding the first token (ms)
|
||||||
mean_tpot_ms: float
|
mean_tpot_ms: float
|
||||||
median_tpot_ms: float
|
median_tpot_ms: float
|
||||||
std_tpot_ms: float
|
std_tpot_ms: float
|
||||||
|
p90_tpot_ms: float
|
||||||
|
p95_tpot_ms: float
|
||||||
p99_tpot_ms: float
|
p99_tpot_ms: float
|
||||||
|
|
||||||
|
# ITL - Inter-Token Latency (ms)
|
||||||
mean_itl_ms: float
|
mean_itl_ms: float
|
||||||
median_itl_ms: float
|
median_itl_ms: float
|
||||||
std_itl_ms: float
|
std_itl_ms: float
|
||||||
|
p90_itl_ms: float
|
||||||
p95_itl_ms: float
|
p95_itl_ms: float
|
||||||
p99_itl_ms: float
|
p99_itl_ms: float
|
||||||
max_itl_ms: float
|
max_itl_ms: float
|
||||||
|
|
||||||
|
# E2E - End-to-End request latency (ms)
|
||||||
mean_e2e_latency_ms: float
|
mean_e2e_latency_ms: float
|
||||||
median_e2e_latency_ms: float
|
median_e2e_latency_ms: float
|
||||||
std_e2e_latency_ms: float
|
std_e2e_latency_ms: float
|
||||||
p90_e2e_latency_ms: float
|
p90_e2e_latency_ms: float
|
||||||
|
p95_e2e_latency_ms: float
|
||||||
p99_e2e_latency_ms: float
|
p99_e2e_latency_ms: float
|
||||||
|
|
||||||
|
# Concurrency and peak metrics
|
||||||
concurrency: float
|
concurrency: float
|
||||||
max_output_tokens_per_s: float = 0.0
|
max_output_tokens_per_s: float = 0.0
|
||||||
max_concurrent_requests: int = 0
|
max_concurrent_requests: int = 0
|
||||||
@@ -1116,14 +1135,19 @@ def calculate_metrics(
|
|||||||
* 1000, # ttfts is empty if streaming is not supported by backend
|
* 1000, # ttfts is empty if streaming is not supported by backend
|
||||||
median_ttft_ms=np.median(ttfts or 0) * 1000,
|
median_ttft_ms=np.median(ttfts or 0) * 1000,
|
||||||
std_ttft_ms=np.std(ttfts or 0) * 1000,
|
std_ttft_ms=np.std(ttfts or 0) * 1000,
|
||||||
|
p90_ttft_ms=np.percentile(ttfts or 0, 90) * 1000,
|
||||||
|
p95_ttft_ms=np.percentile(ttfts or 0, 95) * 1000,
|
||||||
p99_ttft_ms=np.percentile(ttfts or 0, 99) * 1000,
|
p99_ttft_ms=np.percentile(ttfts or 0, 99) * 1000,
|
||||||
mean_tpot_ms=np.mean(tpots or 0) * 1000,
|
mean_tpot_ms=np.mean(tpots or 0) * 1000,
|
||||||
median_tpot_ms=np.median(tpots or 0) * 1000,
|
median_tpot_ms=np.median(tpots or 0) * 1000,
|
||||||
std_tpot_ms=np.std(tpots or 0) * 1000,
|
std_tpot_ms=np.std(tpots or 0) * 1000,
|
||||||
|
p90_tpot_ms=np.percentile(tpots or 0, 90) * 1000,
|
||||||
|
p95_tpot_ms=np.percentile(tpots or 0, 95) * 1000,
|
||||||
p99_tpot_ms=np.percentile(tpots or 0, 99) * 1000,
|
p99_tpot_ms=np.percentile(tpots or 0, 99) * 1000,
|
||||||
mean_itl_ms=np.mean(itls or 0) * 1000,
|
mean_itl_ms=np.mean(itls or 0) * 1000,
|
||||||
median_itl_ms=np.median(itls or 0) * 1000,
|
median_itl_ms=np.median(itls or 0) * 1000,
|
||||||
std_itl_ms=np.std(itls or 0) * 1000,
|
std_itl_ms=np.std(itls or 0) * 1000,
|
||||||
|
p90_itl_ms=np.percentile(itls or 0, 90) * 1000,
|
||||||
p95_itl_ms=np.percentile(itls or 0, 95) * 1000,
|
p95_itl_ms=np.percentile(itls or 0, 95) * 1000,
|
||||||
p99_itl_ms=np.percentile(itls or 0, 99) * 1000,
|
p99_itl_ms=np.percentile(itls or 0, 99) * 1000,
|
||||||
max_itl_ms=np.max(itls or 0) * 1000,
|
max_itl_ms=np.max(itls or 0) * 1000,
|
||||||
@@ -1131,6 +1155,7 @@ def calculate_metrics(
|
|||||||
median_e2e_latency_ms=np.median(e2e_latencies) * 1000,
|
median_e2e_latency_ms=np.median(e2e_latencies) * 1000,
|
||||||
std_e2e_latency_ms=np.std(e2e_latencies) * 1000,
|
std_e2e_latency_ms=np.std(e2e_latencies) * 1000,
|
||||||
p90_e2e_latency_ms=np.percentile(e2e_latencies, 90) * 1000,
|
p90_e2e_latency_ms=np.percentile(e2e_latencies, 90) * 1000,
|
||||||
|
p95_e2e_latency_ms=np.percentile(e2e_latencies, 95) * 1000,
|
||||||
p99_e2e_latency_ms=np.percentile(e2e_latencies, 99) * 1000,
|
p99_e2e_latency_ms=np.percentile(e2e_latencies, 99) * 1000,
|
||||||
concurrency=np.sum(e2e_latencies) / dur_s,
|
concurrency=np.sum(e2e_latencies) / dur_s,
|
||||||
max_output_tokens_per_s=max_output_tokens_per_s,
|
max_output_tokens_per_s=max_output_tokens_per_s,
|
||||||
@@ -1553,6 +1578,9 @@ async def benchmark(
|
|||||||
print(
|
print(
|
||||||
"{:<40} {:<10.2f}".format("P90 E2E Latency (ms):", metrics.p90_e2e_latency_ms)
|
"{:<40} {:<10.2f}".format("P90 E2E Latency (ms):", metrics.p90_e2e_latency_ms)
|
||||||
)
|
)
|
||||||
|
print(
|
||||||
|
"{:<40} {:<10.2f}".format("P95 E2E Latency (ms):", metrics.p95_e2e_latency_ms)
|
||||||
|
)
|
||||||
print(
|
print(
|
||||||
"{:<40} {:<10.2f}".format("P99 E2E Latency (ms):", metrics.p99_e2e_latency_ms)
|
"{:<40} {:<10.2f}".format("P99 E2E Latency (ms):", metrics.p99_e2e_latency_ms)
|
||||||
)
|
)
|
||||||
@@ -1560,6 +1588,8 @@ async def benchmark(
|
|||||||
print("{s:{c}^{n}}".format(s="Time to First Token", n=50, c="-"))
|
print("{s:{c}^{n}}".format(s="Time to First Token", n=50, c="-"))
|
||||||
print("{:<40} {:<10.2f}".format("Mean TTFT (ms):", metrics.mean_ttft_ms))
|
print("{:<40} {:<10.2f}".format("Mean TTFT (ms):", metrics.mean_ttft_ms))
|
||||||
print("{:<40} {:<10.2f}".format("Median TTFT (ms):", metrics.median_ttft_ms))
|
print("{:<40} {:<10.2f}".format("Median TTFT (ms):", metrics.median_ttft_ms))
|
||||||
|
print("{:<40} {:<10.2f}".format("P90 TTFT (ms):", metrics.p90_ttft_ms))
|
||||||
|
print("{:<40} {:<10.2f}".format("P95 TTFT (ms):", metrics.p95_ttft_ms))
|
||||||
print("{:<40} {:<10.2f}".format("P99 TTFT (ms):", metrics.p99_ttft_ms))
|
print("{:<40} {:<10.2f}".format("P99 TTFT (ms):", metrics.p99_ttft_ms))
|
||||||
print(
|
print(
|
||||||
"{s:{c}^{n}}".format(
|
"{s:{c}^{n}}".format(
|
||||||
@@ -1568,10 +1598,13 @@ async def benchmark(
|
|||||||
)
|
)
|
||||||
print("{:<40} {:<10.2f}".format("Mean TPOT (ms):", metrics.mean_tpot_ms))
|
print("{:<40} {:<10.2f}".format("Mean TPOT (ms):", metrics.mean_tpot_ms))
|
||||||
print("{:<40} {:<10.2f}".format("Median TPOT (ms):", metrics.median_tpot_ms))
|
print("{:<40} {:<10.2f}".format("Median TPOT (ms):", metrics.median_tpot_ms))
|
||||||
|
print("{:<40} {:<10.2f}".format("P90 TPOT (ms):", metrics.p90_tpot_ms))
|
||||||
|
print("{:<40} {:<10.2f}".format("P95 TPOT (ms):", metrics.p95_tpot_ms))
|
||||||
print("{:<40} {:<10.2f}".format("P99 TPOT (ms):", metrics.p99_tpot_ms))
|
print("{:<40} {:<10.2f}".format("P99 TPOT (ms):", metrics.p99_tpot_ms))
|
||||||
print("{s:{c}^{n}}".format(s="Inter-Token Latency", n=50, c="-"))
|
print("{s:{c}^{n}}".format(s="Inter-Token Latency", n=50, c="-"))
|
||||||
print("{:<40} {:<10.2f}".format("Mean ITL (ms):", metrics.mean_itl_ms))
|
print("{:<40} {:<10.2f}".format("Mean ITL (ms):", metrics.mean_itl_ms))
|
||||||
print("{:<40} {:<10.2f}".format("Median ITL (ms):", metrics.median_itl_ms))
|
print("{:<40} {:<10.2f}".format("Median ITL (ms):", metrics.median_itl_ms))
|
||||||
|
print("{:<40} {:<10.2f}".format("P90 ITL (ms):", metrics.p90_itl_ms))
|
||||||
print("{:<40} {:<10.2f}".format("P95 ITL (ms):", metrics.p95_itl_ms))
|
print("{:<40} {:<10.2f}".format("P95 ITL (ms):", metrics.p95_itl_ms))
|
||||||
print("{:<40} {:<10.2f}".format("P99 ITL (ms):", metrics.p99_itl_ms))
|
print("{:<40} {:<10.2f}".format("P99 ITL (ms):", metrics.p99_itl_ms))
|
||||||
print("{:<40} {:<10.2f}".format("Max ITL (ms):", metrics.max_itl_ms))
|
print("{:<40} {:<10.2f}".format("Max ITL (ms):", metrics.max_itl_ms))
|
||||||
@@ -1614,18 +1647,24 @@ async def benchmark(
|
|||||||
"median_e2e_latency_ms": metrics.median_e2e_latency_ms,
|
"median_e2e_latency_ms": metrics.median_e2e_latency_ms,
|
||||||
"std_e2e_latency_ms": metrics.std_e2e_latency_ms,
|
"std_e2e_latency_ms": metrics.std_e2e_latency_ms,
|
||||||
"p90_e2e_latency_ms": metrics.p90_e2e_latency_ms,
|
"p90_e2e_latency_ms": metrics.p90_e2e_latency_ms,
|
||||||
|
"p95_e2e_latency_ms": metrics.p95_e2e_latency_ms,
|
||||||
"p99_e2e_latency_ms": metrics.p99_e2e_latency_ms,
|
"p99_e2e_latency_ms": metrics.p99_e2e_latency_ms,
|
||||||
"mean_ttft_ms": metrics.mean_ttft_ms,
|
"mean_ttft_ms": metrics.mean_ttft_ms,
|
||||||
"median_ttft_ms": metrics.median_ttft_ms,
|
"median_ttft_ms": metrics.median_ttft_ms,
|
||||||
"std_ttft_ms": metrics.std_ttft_ms,
|
"std_ttft_ms": metrics.std_ttft_ms,
|
||||||
|
"p90_ttft_ms": metrics.p90_ttft_ms,
|
||||||
|
"p95_ttft_ms": metrics.p95_ttft_ms,
|
||||||
"p99_ttft_ms": metrics.p99_ttft_ms,
|
"p99_ttft_ms": metrics.p99_ttft_ms,
|
||||||
"mean_tpot_ms": metrics.mean_tpot_ms,
|
"mean_tpot_ms": metrics.mean_tpot_ms,
|
||||||
"median_tpot_ms": metrics.median_tpot_ms,
|
"median_tpot_ms": metrics.median_tpot_ms,
|
||||||
"std_tpot_ms": metrics.std_tpot_ms,
|
"std_tpot_ms": metrics.std_tpot_ms,
|
||||||
|
"p90_tpot_ms": metrics.p90_tpot_ms,
|
||||||
|
"p95_tpot_ms": metrics.p95_tpot_ms,
|
||||||
"p99_tpot_ms": metrics.p99_tpot_ms,
|
"p99_tpot_ms": metrics.p99_tpot_ms,
|
||||||
"mean_itl_ms": metrics.mean_itl_ms,
|
"mean_itl_ms": metrics.mean_itl_ms,
|
||||||
"median_itl_ms": metrics.median_itl_ms,
|
"median_itl_ms": metrics.median_itl_ms,
|
||||||
"std_itl_ms": metrics.std_itl_ms,
|
"std_itl_ms": metrics.std_itl_ms,
|
||||||
|
"p90_itl_ms": metrics.p90_itl_ms,
|
||||||
"p95_itl_ms": metrics.p95_itl_ms,
|
"p95_itl_ms": metrics.p95_itl_ms,
|
||||||
"p99_itl_ms": metrics.p99_itl_ms,
|
"p99_itl_ms": metrics.p99_itl_ms,
|
||||||
"concurrency": metrics.concurrency,
|
"concurrency": metrics.concurrency,
|
||||||
|
|||||||
Reference in New Issue
Block a user