Skip to content

Commit 9b0c7d6

Browse files
committed
added query accuracy comparison between baseline and asap
1 parent 7e1983b commit 9b0c7d6

1 file changed

Lines changed: 97 additions & 10 deletions

File tree

‎asap-tools/execution-utilities/benchmark/run_benchmark.py‎

Lines changed: 97 additions & 10 deletions
Original file line numberDiff line numberDiff line change
@@ -178,6 +178,7 @@ def run_query(
178178
return (
179179
latency_ms,
180180
None,
181+
0,
181182
f"HTTP {response.status_code}: {response.text[:200]}",
182183
)
183184
except requests.Timeout:
@@ -300,7 +301,7 @@ def run_benchmark(
300301
)
301302
plot_latencies.append(0.0)
302303
else:
303-
preview = last_result.replace("\n", " | ")[:200] if last_result else ""
304+
preview = last_result.replace("\n", " | ") if last_result else ""
304305
latencies_ok.append(latency_ms)
305306
plot_latencies.append(latency_ms)
306307
print(f"{latency_ms:.2f}ms ({last_row_count} rows)")
@@ -341,18 +342,56 @@ def _plot_single(latencies: List[float], mode: str, out_path: Path):
341342
print(f"Plot saved to {out_path}")
342343

343344

344-
def _plot_comparison(asap_csv: Path, baseline_csv: Path, out_path: Path):
345-
"""Two-panel comparison plot: per-query bars + speedup bars.
345+
def _parse_result_values(result_full: str) -> List[float]:
346+
"""Extract numeric values from a pipe-separated result_full string."""
347+
if not result_full:
348+
return []
349+
values = []
350+
for part in result_full.split(" | "):
351+
part = part.strip()
352+
if not part:
353+
continue
354+
cols = part.split("\t")
355+
try:
356+
values.append(float(cols[-1]))
357+
except (ValueError, IndexError):
358+
continue
359+
return values
360+
361+
362+
def _compute_result_error(
363+
baseline_values: List[float], asap_values: List[float]
364+
) -> Optional[float]:
365+
"""Mean absolute relative error between two sorted result sets."""
366+
if not baseline_values or not asap_values:
367+
return None
368+
b = sorted(baseline_values)
369+
a = sorted(asap_values)
370+
n = min(len(b), len(a))
371+
if n == 0:
372+
return None
373+
b, a = b[:n], a[:n]
374+
errors = []
375+
for bv, av in zip(b, a):
376+
if bv == 0:
377+
errors.append(0.0 if av == 0 else abs(av))
378+
else:
379+
errors.append(abs(av - bv) / abs(bv))
380+
return sum(errors) / len(errors)
346381

347-
Adapted from asap_query_latency/plot_latency.py.
348-
"""
382+
383+
def _plot_comparison(asap_csv: Path, baseline_csv: Path, out_path: Path):
384+
"""Three-panel comparison: latency bars, speedup, and result accuracy."""
349385

350386
def _load(path):
351387
rows = {}
352388
with open(path) as f:
353389
for row in csv.DictReader(f):
354390
if not row["error"]:
355-
rows[row["query_id"]] = float(row["latency_ms"])
391+
rows[row["query_id"]] = {
392+
"latency": float(row["latency_ms"]),
393+
"result": row.get("result_full", ""),
394+
}
356395
return rows
357396

358397
asap = _load(asap_csv)
@@ -363,13 +402,28 @@ def _load(path):
363402
return
364403

365404
x = np.arange(len(qids))
366-
a_vals = [asap[q] for q in qids]
367-
b_vals = [base[q] for q in qids]
405+
a_vals = [asap[q]["latency"] for q in qids]
406+
b_vals = [base[q]["latency"] for q in qids]
368407
speedup = [b / a if a > 0 else 0 for a, b in zip(a_vals, b_vals)]
369408

370-
fig, (ax1, ax2) = plt.subplots(
371-
2, 1, figsize=(14, 7), gridspec_kw={"height_ratios": [3, 1]}
409+
errors_pct = []
410+
for q in qids:
411+
b_results = _parse_result_values(base[q]["result"])
412+
a_results = _parse_result_values(asap[q]["result"])
413+
err = _compute_result_error(b_results, a_results)
414+
errors_pct.append((err or 0.0) * 100)
415+
416+
has_accuracy = any(e > 0 for e in errors_pct)
417+
n_panels = 3 if has_accuracy else 2
418+
ratios = [3, 1, 1.5] if has_accuracy else [3, 1]
419+
420+
fig, axes = plt.subplots(
421+
n_panels,
422+
1,
423+
figsize=(14, 4 + 3 * n_panels),
424+
gridspec_kw={"height_ratios": ratios},
372425
)
426+
ax1, ax2 = axes[0], axes[1]
373427

374428
w = 0.4
375429
ax1.bar(x - w / 2, b_vals, w, label="Baseline", color="#f4a460")
@@ -398,11 +452,44 @@ def _load(path):
398452
ax2.legend(fontsize=8)
399453
ax2.set_xlim(-0.6, len(qids) - 0.4)
400454

455+
if has_accuracy:
456+
ax3 = axes[2]
457+
colors = [
458+
"#d9534f" if e > 10 else "#f0ad4e" if e > 5 else "#5cb85c"
459+
for e in errors_pct
460+
]
461+
ax3.bar(
462+
x, errors_pct, color=colors, width=0.7, edgecolor="black", linewidth=0.3
463+
)
464+
mean_err = np.mean(errors_pct)
465+
ax3.axhline(
466+
mean_err,
467+
color="red",
468+
linewidth=1,
469+
linestyle="--",
470+
label=f"mean {mean_err:.2f}%",
471+
)
472+
ax3.set_xticks(x)
473+
ax3.set_xticklabels(qids, rotation=90, fontsize=7)
474+
ax3.set_ylabel("Relative Error (%)")
475+
ax3.set_title("Result accuracy: ASAP estimate vs baseline exact answer")
476+
ax3.legend(fontsize=8)
477+
ax3.set_xlim(-0.6, len(qids) - 0.4)
478+
401479
plt.tight_layout()
402480
plt.savefig(out_path, dpi=150)
403481
plt.close()
404482
print(f"Comparison plot saved to {out_path}")
405483

484+
if has_accuracy:
485+
s = sorted(errors_pct)
486+
n = len(s)
487+
print(
488+
f"Result error: mean={np.mean(s):.2f}% "
489+
f"p50={s[int(n*0.50)]:.2f}% p95={s[int(n*0.95)]:.2f}% "
490+
f"max={s[-1]:.2f}%"
491+
)
492+
406493

407494
# ---------------------------------------------------------------------------
408495
# Main

0 commit comments

Comments
 (0)