@@ -178,6 +178,7 @@ def run_query(
178178 return (
179179 latency_ms ,
180180 None ,
181+ 0 ,
181182 f"HTTP { response .status_code } : { response .text [:200 ]} " ,
182183 )
183184 except requests .Timeout :
@@ -300,7 +301,7 @@ def run_benchmark(
300301 )
301302 plot_latencies .append (0.0 )
302303 else :
303- preview = last_result .replace ("\n " , " | " )[: 200 ] if last_result else ""
304+ preview = last_result .replace ("\n " , " | " ) if last_result else ""
304305 latencies_ok .append (latency_ms )
305306 plot_latencies .append (latency_ms )
306307 print (f"{ latency_ms :.2f} ms ({ last_row_count } rows)" )
@@ -341,18 +342,56 @@ def _plot_single(latencies: List[float], mode: str, out_path: Path):
341342 print (f"Plot saved to { out_path } " )
342343
343344
344- def _plot_comparison (asap_csv : Path , baseline_csv : Path , out_path : Path ):
345- """Two-panel comparison plot: per-query bars + speedup bars.
345+ def _parse_result_values (result_full : str ) -> List [float ]:
346+ """Extract numeric values from a pipe-separated result_full string."""
347+ if not result_full :
348+ return []
349+ values = []
350+ for part in result_full .split (" | " ):
351+ part = part .strip ()
352+ if not part :
353+ continue
354+ cols = part .split ("\t " )
355+ try :
356+ values .append (float (cols [- 1 ]))
357+ except (ValueError , IndexError ):
358+ continue
359+ return values
360+
361+
362+ def _compute_result_error (
363+ baseline_values : List [float ], asap_values : List [float ]
364+ ) -> Optional [float ]:
365+ """Mean absolute relative error between two sorted result sets."""
366+ if not baseline_values or not asap_values :
367+ return None
368+ b = sorted (baseline_values )
369+ a = sorted (asap_values )
370+ n = min (len (b ), len (a ))
371+ if n == 0 :
372+ return None
373+ b , a = b [:n ], a [:n ]
374+ errors = []
375+ for bv , av in zip (b , a ):
376+ if bv == 0 :
377+ errors .append (0.0 if av == 0 else abs (av ))
378+ else :
379+ errors .append (abs (av - bv ) / abs (bv ))
380+ return sum (errors ) / len (errors )
346381
347- Adapted from asap_query_latency/plot_latency.py.
348- """
382+
383+ def _plot_comparison (asap_csv : Path , baseline_csv : Path , out_path : Path ):
384+ """Three-panel comparison: latency bars, speedup, and result accuracy."""
349385
350386 def _load (path ):
351387 rows = {}
352388 with open (path ) as f :
353389 for row in csv .DictReader (f ):
354390 if not row ["error" ]:
355- rows [row ["query_id" ]] = float (row ["latency_ms" ])
391+ rows [row ["query_id" ]] = {
392+ "latency" : float (row ["latency_ms" ]),
393+ "result" : row .get ("result_full" , "" ),
394+ }
356395 return rows
357396
358397 asap = _load (asap_csv )
@@ -363,13 +402,28 @@ def _load(path):
363402 return
364403
365404 x = np .arange (len (qids ))
366- a_vals = [asap [q ] for q in qids ]
367- b_vals = [base [q ] for q in qids ]
405+ a_vals = [asap [q ][ "latency" ] for q in qids ]
406+ b_vals = [base [q ][ "latency" ] for q in qids ]
368407 speedup = [b / a if a > 0 else 0 for a , b in zip (a_vals , b_vals )]
369408
370- fig , (ax1 , ax2 ) = plt .subplots (
371- 2 , 1 , figsize = (14 , 7 ), gridspec_kw = {"height_ratios" : [3 , 1 ]}
409+ errors_pct = []
410+ for q in qids :
411+ b_results = _parse_result_values (base [q ]["result" ])
412+ a_results = _parse_result_values (asap [q ]["result" ])
413+ err = _compute_result_error (b_results , a_results )
414+ errors_pct .append ((err or 0.0 ) * 100 )
415+
416+ has_accuracy = any (e > 0 for e in errors_pct )
417+ n_panels = 3 if has_accuracy else 2
418+ ratios = [3 , 1 , 1.5 ] if has_accuracy else [3 , 1 ]
419+
420+ fig , axes = plt .subplots (
421+ n_panels ,
422+ 1 ,
423+ figsize = (14 , 4 + 3 * n_panels ),
424+ gridspec_kw = {"height_ratios" : ratios },
372425 )
426+ ax1 , ax2 = axes [0 ], axes [1 ]
373427
374428 w = 0.4
375429 ax1 .bar (x - w / 2 , b_vals , w , label = "Baseline" , color = "#f4a460" )
@@ -398,11 +452,44 @@ def _load(path):
398452 ax2 .legend (fontsize = 8 )
399453 ax2 .set_xlim (- 0.6 , len (qids ) - 0.4 )
400454
455+ if has_accuracy :
456+ ax3 = axes [2 ]
457+ colors = [
458+ "#d9534f" if e > 10 else "#f0ad4e" if e > 5 else "#5cb85c"
459+ for e in errors_pct
460+ ]
461+ ax3 .bar (
462+ x , errors_pct , color = colors , width = 0.7 , edgecolor = "black" , linewidth = 0.3
463+ )
464+ mean_err = np .mean (errors_pct )
465+ ax3 .axhline (
466+ mean_err ,
467+ color = "red" ,
468+ linewidth = 1 ,
469+ linestyle = "--" ,
470+ label = f"mean { mean_err :.2f} %" ,
471+ )
472+ ax3 .set_xticks (x )
473+ ax3 .set_xticklabels (qids , rotation = 90 , fontsize = 7 )
474+ ax3 .set_ylabel ("Relative Error (%)" )
475+ ax3 .set_title ("Result accuracy: ASAP estimate vs baseline exact answer" )
476+ ax3 .legend (fontsize = 8 )
477+ ax3 .set_xlim (- 0.6 , len (qids ) - 0.4 )
478+
401479 plt .tight_layout ()
402480 plt .savefig (out_path , dpi = 150 )
403481 plt .close ()
404482 print (f"Comparison plot saved to { out_path } " )
405483
484+ if has_accuracy :
485+ s = sorted (errors_pct )
486+ n = len (s )
487+ print (
488+ f"Result error: mean={ np .mean (s ):.2f} % "
489+ f"p50={ s [int (n * 0.50 )]:.2f} % p95={ s [int (n * 0.95 )]:.2f} % "
490+ f"max={ s [- 1 ]:.2f} %"
491+ )
492+
406493
407494# ---------------------------------------------------------------------------
408495# Main
0 commit comments