diff --git a/evalscope/perf/utils/db_util.py b/evalscope/perf/utils/db_util.py index b7fa393b1..a7f71a16c 100644 --- a/evalscope/perf/utils/db_util.py +++ b/evalscope/perf/utils/db_util.py @@ -295,6 +295,9 @@ def get_percentile_results( all_percentiles = [0] + percentiles + [100] transposed: Dict[str, list] = {PercentileMetrics.PERCENTILES: percentile_labels} for metric_name, data in metrics.items(): + # NaN marks a sample with no value (e.g. Decode (tok/s) for a reply with no decode phase). + # Leave it out: sort() can't order NaN, which would scramble every percentile of the column. + data = [v for v in data if not (isinstance(v, float) and math.isnan(v))] metric_percentiles = calculate_percentiles(data, all_percentiles) transposed[metric_name] = [metric_percentiles[p] for p in all_percentiles] diff --git a/tests/perf/test_stream_metrics.py b/tests/perf/test_stream_metrics.py index b1e188784..53a910325 100644 --- a/tests/perf/test_stream_metrics.py +++ b/tests/perf/test_stream_metrics.py @@ -293,6 +293,29 @@ def test_min_row_reports_best_case_latency(self): max_row = next(r for r in rows if r['Percentiles'] == 'max') self.assertLessEqual(min_row['TPOT (ms)'], max_row['TPOT (ms)']) + def test_decode_throughput_percentiles_skip_replies_without_decode(self): + # A one-token reply has TPOT 0 and no decode speed; it used to enter the + # Decode (tok/s) column as NaN, which sort() can't order. + db = tempfile.mktemp(suffix='.db') + con = sqlite3.connect(db) + cur = con.cursor() + create_result_table(cur) + tpots = [0.02, 0.0, 0.01, 0.04, 0.03, 0.025] + for tpot in tpots: + bd = _make(tpot=tpot, completion_tokens=1 if tpot == 0.0 else 50, is_stream=True) + insert_benchmark_data(cur, bd) + con.commit() + con.close() + try: + rows = get_percentile_results(db, api_type='openai').to_list() + decode = {r['Percentiles']: r[PercentileMetrics.DECODE_THROUGHPUT] for r in rows} + # Decode speeds of the five replies that have one: 25, 33.33, 40, 50 and 100 tok/s. + self.assertEqual(decode['min'], 25.0) + self.assertEqual(decode['50%'], 40.0) + self.assertEqual(decode['max'], 100.0) + finally: + os.unlink(db) + def test_pure_non_stream_percentiles_fall_back_to_all_rows(self): # Pure non-stream run: no stream rows, so streaming metrics fall back to # all rows (backward compatible) instead of producing NaN.