Skip to content

Commit 1349023

Browse files
zzylolclaude
andcommitted
Add accuracy/performance CI and benchmarks infrastructure
- accuracy_performance.yml: full e2e eval — builds ASAP service images once (planner-rs, summary-ingest, query-engine) with sha tag, spins up quickstart stack, waits for Arroyo pipeline + sketch ingestion, runs PromQL suite against Prometheus (baseline) and ASAPQuery, then compares accuracy and latency - benchmarks/docker-compose.yml: compose override replacing GHCR images with ASAP_IMAGE_TAG so CI tests latest committed code - benchmarks/queries/promql_suite.json: 14-query fixed suite covering avg/sum/max/min/quantile at p50/p90/p95/p99 with/without grouping - benchmarks/scripts/: compare.py, run_baseline.py, run_asap.py, wait_for_stack.sh, ingest_wait.sh - asap-summary-ingest/Dockerfile: switch FROM to ghcr.io base image so builds work without a local sketchdb-base image Pass/fail policy: no query failures; ASAP-native relative error >5% warns (does not fail); latency regressions are warn-only on ephemeral GH runners. Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
1 parent 03407a5 commit 1349023

11 files changed

Lines changed: 863 additions & 1 deletion

File tree

Lines changed: 162 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,162 @@
1+
name: PR Evaluation
2+
3+
# NOTE: GitHub-hosted runners are noisy. Latency numbers are indicative only.
4+
# For precise benchmarks, register a self-hosted runner once asap-tools infra
5+
# is decoupled from Cloudlab. See PDF eval guide Phase 3.
6+
7+
on:
8+
pull_request:
9+
branches:
10+
- main
11+
paths:
12+
- 'asap-query-engine/**'
13+
- 'asap-planner-rs/**'
14+
- 'asap-summary-ingest/**'
15+
- 'asap-quickstart/**'
16+
- '.github/workflows/accuracy_performance.yml'
17+
- 'benchmarks/**'
18+
workflow_dispatch:
19+
20+
permissions:
21+
contents: read
22+
packages: write
23+
pull-requests: write
24+
25+
jobs:
26+
# ---------------------------------------------------------------------------
27+
# Job 1: build images once from branch code and push with a SHA-based tag.
28+
# All downstream jobs pull these images instead of rebuilding.
29+
# ---------------------------------------------------------------------------
30+
build:
31+
name: Build CI images
32+
runs-on: ubuntu-latest
33+
outputs:
34+
image-tag: ${{ steps.tag.outputs.value }}
35+
36+
steps:
37+
- name: Checkout repository
38+
uses: actions/checkout@v4
39+
40+
- name: Set up Docker Buildx
41+
uses: docker/setup-buildx-action@v3
42+
43+
- name: Log in to GHCR
44+
uses: docker/login-action@v3
45+
with:
46+
registry: ghcr.io
47+
username: ${{ github.repository_owner }}
48+
password: ${{ secrets.GITHUB_TOKEN }}
49+
50+
- name: Compute image tag
51+
id: tag
52+
run: echo "value=sha-$(git rev-parse --short HEAD)" >> $GITHUB_OUTPUT
53+
54+
- name: Build and push asap-planner-rs
55+
uses: docker/build-push-action@v6
56+
with:
57+
context: .
58+
file: asap-planner-rs/Dockerfile
59+
push: true
60+
tags: ghcr.io/projectasap/asap-planner-rs:${{ steps.tag.outputs.value }}
61+
cache-from: type=registry,ref=ghcr.io/projectasap/asap-planner-rs:buildcache
62+
cache-to: type=registry,ref=ghcr.io/projectasap/asap-planner-rs:buildcache,mode=max
63+
64+
- name: Build and push asap-summary-ingest
65+
uses: docker/build-push-action@v6
66+
with:
67+
context: asap-summary-ingest
68+
file: asap-summary-ingest/Dockerfile
69+
push: true
70+
tags: ghcr.io/projectasap/asap-summary-ingest:${{ steps.tag.outputs.value }}
71+
72+
- name: Build and push asap-query-engine
73+
uses: docker/build-push-action@v6
74+
with:
75+
context: .
76+
file: asap-query-engine/Dockerfile
77+
push: true
78+
tags: ghcr.io/projectasap/asap-query-engine:${{ steps.tag.outputs.value }}
79+
cache-from: type=registry,ref=ghcr.io/projectasap/asap-query-engine:buildcache
80+
cache-to: type=registry,ref=ghcr.io/projectasap/asap-query-engine:buildcache,mode=max
81+
82+
# ---------------------------------------------------------------------------
83+
# Job 2: pull the images built above, deploy the full stack, and evaluate.
84+
# ---------------------------------------------------------------------------
85+
eval:
86+
name: Full-stack PR evaluation
87+
needs: build
88+
runs-on: ubuntu-latest
89+
timeout-minutes: 60
90+
env:
91+
ASAP_IMAGE_TAG: ${{ needs.build.outputs.image-tag }}
92+
93+
steps:
94+
- name: Checkout repository
95+
uses: actions/checkout@v4
96+
97+
- name: Log in to GHCR
98+
uses: docker/login-action@v3
99+
with:
100+
registry: ghcr.io
101+
username: ${{ github.repository_owner }}
102+
password: ${{ secrets.GITHUB_TOKEN }}
103+
104+
- name: Pull and start full stack
105+
run: |
106+
docker compose \
107+
-f asap-quickstart/docker-compose.yml \
108+
-f benchmarks/docker-compose.yml \
109+
up -d
110+
111+
- name: Show running containers
112+
run: |
113+
docker compose \
114+
-f asap-quickstart/docker-compose.yml \
115+
-f benchmarks/docker-compose.yml \
116+
ps
117+
118+
- name: Wait for all services to be healthy
119+
run: bash benchmarks/scripts/wait_for_stack.sh
120+
121+
- name: Wait for pipeline and data ingestion
122+
run: bash benchmarks/scripts/ingest_wait.sh
123+
124+
- name: Set up Python 3.11
125+
uses: actions/setup-python@v5
126+
with:
127+
python-version: '3.11'
128+
129+
- name: Install Python dependencies
130+
run: pip install requests
131+
132+
- name: Run baseline queries (Prometheus)
133+
run: python benchmarks/scripts/run_baseline.py
134+
135+
- name: Run ASAP queries (query engine)
136+
run: python benchmarks/scripts/run_asap.py
137+
138+
- name: Compare results and evaluate
139+
run: python benchmarks/scripts/compare.py
140+
141+
- name: Upload evaluation reports
142+
if: always()
143+
uses: actions/upload-artifact@v4
144+
with:
145+
name: eval-reports-${{ github.run_id }}
146+
path: benchmarks/reports/
147+
148+
- name: Print docker logs on failure
149+
if: failure()
150+
run: |
151+
docker compose \
152+
-f asap-quickstart/docker-compose.yml \
153+
-f benchmarks/docker-compose.yml \
154+
logs --no-color
155+
156+
- name: Teardown stack
157+
if: always()
158+
run: |
159+
docker compose \
160+
-f asap-quickstart/docker-compose.yml \
161+
-f benchmarks/docker-compose.yml \
162+
down -v

‎asap-summary-ingest/Dockerfile‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,4 @@
1-
FROM sketchdb-base:latest
1+
FROM ghcr.io/projectasap/asap-base:latest
22

33
LABEL maintainer="SketchDB Team"
44
LABEL description="ArroyoSketch pipeline configuration service"

‎benchmarks/docker-compose.yml‎

Lines changed: 20 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,20 @@
1+
# CI image override: replaces quickstart's pinned release images with images
2+
# built from the current branch. Intended for use as a Compose override:
3+
#
4+
# ASAP_IMAGE_TAG=sha-<short-sha> docker compose \
5+
# --project-directory . \
6+
# -f asap-quickstart/docker-compose.yml \
7+
# -f benchmarks/docker-compose.yml \
8+
# up -d
9+
#
10+
# ASAP_IMAGE_TAG is set automatically by the 'build' job in accuracy_performance.yml.
11+
12+
services:
13+
asap-planner-rs:
14+
image: ghcr.io/projectasap/asap-planner-rs:${ASAP_IMAGE_TAG}
15+
16+
asap-summary-ingest:
17+
image: ghcr.io/projectasap/asap-summary-ingest:${ASAP_IMAGE_TAG}
18+
19+
queryengine:
20+
image: ghcr.io/projectasap/asap-query-engine:${ASAP_IMAGE_TAG}

‎benchmarks/golden/.gitkeep‎

Whitespace-only changes.
Lines changed: 18 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,18 @@
1+
{
2+
"queries": [
3+
{"id": "avg_all", "expr": "avg(sensor_reading)", "asap_native": false},
4+
{"id": "sum_all", "expr": "sum(sensor_reading)", "asap_native": false},
5+
{"id": "max_all", "expr": "max(sensor_reading)", "asap_native": false},
6+
{"id": "min_all", "expr": "min(sensor_reading)", "asap_native": false},
7+
{"id": "q50_all", "expr": "quantile(0.50, sensor_reading)", "asap_native": true},
8+
{"id": "q90_all", "expr": "quantile(0.90, sensor_reading)", "asap_native": true},
9+
{"id": "q95_all", "expr": "quantile(0.95, sensor_reading)", "asap_native": true},
10+
{"id": "q99_all", "expr": "quantile(0.99, sensor_reading)", "asap_native": true},
11+
{"id": "q95_by_pattern", "expr": "quantile by (pattern) (0.95, sensor_reading)", "asap_native": true},
12+
{"id": "q99_by_pattern", "expr": "quantile by (pattern) (0.99, sensor_reading)", "asap_native": true},
13+
{"id": "q50_by_pattern", "expr": "quantile by (pattern) (0.50, sensor_reading)", "asap_native": true},
14+
{"id": "avg_by_pattern", "expr": "avg by (pattern) (sensor_reading)", "asap_native": false},
15+
{"id": "sum_by_region", "expr": "sum by (region) (sensor_reading)", "asap_native": false},
16+
{"id": "max_by_service", "expr": "max by (service) (sensor_reading)", "asap_native": false}
17+
]
18+
}

‎benchmarks/reports/.gitkeep‎

Whitespace-only changes.

0 commit comments

Comments
 (0)