From 04ab28e128c8ec37fa3448b014d986b1156c916b Mon Sep 17 00:00:00 2001 From: zz_y Date: Fri, 17 Jul 2026 22:14:45 -0600 Subject: [PATCH] test(promql): add a small o11y-bench-derived PromQL corpus MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 27 deduplicated queries pulled from grafana/o11y-bench's task-grading rubrics — realistic SRE incident/investigation PromQL (error ratios, burn rate, cache-lag/retry-backlog triage, capacity checks). All 27 lower cleanly; asserts full coverage since the corpus is small and hand-picked rather than a statistical sample. o11y-bench is AGPL-3.0, unlike this repo's other MIT/Apache-2.0 corpora — that's an open question, tracked in #135, not resolved here. --- crates/frontend-promql/Cargo.toml | 4 ++ .../observability/data/o11y_bench_promql.txt | 37 ++++++++++++ .../tests/observability/o11y_bench_promql.rs | 56 +++++++++++++++++++ 3 files changed, 97 insertions(+) create mode 100644 crates/frontend-promql/tests/observability/data/o11y_bench_promql.txt create mode 100644 crates/frontend-promql/tests/observability/o11y_bench_promql.rs diff --git a/crates/frontend-promql/Cargo.toml b/crates/frontend-promql/Cargo.toml index 3bce98cf..422f43bc 100644 --- a/crates/frontend-promql/Cargo.toml +++ b/crates/frontend-promql/Cargo.toml @@ -22,3 +22,7 @@ path = "tests/observability/awesome_prometheus_alerts.rs" [[test]] name = "promql_corpus" path = "tests/observability/promql_corpus.rs" + +[[test]] +name = "o11y_bench_promql" +path = "tests/observability/o11y_bench_promql.rs" diff --git a/crates/frontend-promql/tests/observability/data/o11y_bench_promql.txt b/crates/frontend-promql/tests/observability/data/o11y_bench_promql.txt new file mode 100644 index 00000000..ea2eee98 --- /dev/null +++ b/crates/frontend-promql/tests/observability/data/o11y_bench_promql.txt @@ -0,0 +1,37 @@ +# PromQL query corpus derived from grafana/o11y-bench task-grading rubrics. +# Source: https://github.com/grafana/o11y-bench (tasks-spec/**/*.yaml, `fact.query` +# values where backend == prometheus). Snapshot fetched 2026-07-17; deduplicated. +# License: AGPL-3.0. Vendoring this corpus verbatim from an AGPL-3.0 source into a +# differently-licensed test suite has an open licensing question -- see issue #135. +# Do not treat this file's provenance as cleared; it is tracked, not resolved. +# One real-world PromQL expression per line, pulled from LLM-agent-benchmark grading +# rubrics (SRE-style incident/investigation queries: error ratios, burn rate, cache +# lag, retry backlog, capacity). Used by tests/observability/o11y_bench_promql.rs to +# pin parse + lowering behaviour over real-world queries (totality; no execution). +(sum by (job) (increase(http_requests_total{status=~"5..",job=~"user-service|order-service|payment-service"}[6h]))) / (sum by (job) (increase(http_requests_total{job=~"user-service|order-service|payment-service"}[6h]))) +histogram_quantile(0.95, sum(rate(http_request_duration_seconds_bucket{job="order-service"}[5m])) by (le)) +histogram_quantile(0.99, sum(rate(http_request_duration_seconds_bucket{job="order-service"}[6h])) by (le)) +max((sum by (job) (rate(http_requests_total{status=~"5..",job=~"user-service|order-service|payment-service"}[6h]))) / (sum by (job) (rate(http_requests_total{job=~"user-service|order-service|payment-service"}[6h])))) +max_over_time((sum(rate(http_requests_total{job="order-service",status=~"5.."}[5m])) / sum(rate(http_requests_total{job="order-service"}[5m])))[6h:1m]) +max_over_time(histogram_quantile(0.95, sum(rate(http_request_duration_seconds_bucket{job="user-service"}[5m])) by (le))[12h:1m]) +max_over_time(service_cache_refresh_lag_seconds{job="user-service"}[12h]) +max_over_time(service_retry_queue_depth{job="order-service"}[6h]) +max_over_time(service_retry_queue_depth{job="payment-service"}[6h]) +max_over_time(service_retry_queue_depth{job=~".+"}[6h]) +max_over_time(sum(process_resident_memory_bytes)[6h:1m]) +service_cache_refresh_lag_seconds{job="user-service"} +sort_desc(sum by (job) (rate(process_cpu_seconds_total{job=~".+"}[6h]))) +sum by (job) (increase(http_requests_total{status=~"5..",job=~".+-service"}[24h])) > 0 +sum by (job) (rate(process_cpu_seconds_total{job=~".+"}[6h])) +sum(avg_over_time(process_resident_memory_bytes{job=~".+"}[6h])) +sum(increase(http_requests_total{job="order-service",status=~"5.."}[24h])) / sum(increase(http_requests_total{status=~"5.."}[24h])) +sum(increase(http_requests_total{job="payment-service",status=~"5.."}[1h] offset 6h)) / sum(increase(http_requests_total{job="payment-service"}[1h] offset 6h)) +sum(increase(http_requests_total{job="payment-service",status=~"5.."}[1h])) / sum(increase(http_requests_total{job="payment-service"}[1h])) +sum(process_resident_memory_bytes) +sum(rate(http_requests_total{job="order-service"}[5m] offset 1h)) +sum(rate(http_requests_total{job="order-service"}[5m])) +sum(rate(http_requests_total{job=~"user-service|order-service|payment-service"}[5m])) +sum(rate(http_requests_total{status=~"5..",job=~"payment-service|order-service"}[6h])) / sum(rate(http_requests_total{job=~"payment-service|order-service"}[6h])) +sum(rate(http_requests_total{status=~"5..",job=~"user-service|order-service|payment-service"}[1h])) / sum(rate(http_requests_total{job=~"user-service|order-service|payment-service"}[1h])) +sum(rate(process_cpu_seconds_total[1h])) +sum(up) diff --git a/crates/frontend-promql/tests/observability/o11y_bench_promql.rs b/crates/frontend-promql/tests/observability/o11y_bench_promql.rs new file mode 100644 index 00000000..a753847a --- /dev/null +++ b/crates/frontend-promql/tests/observability/o11y_bench_promql.rs @@ -0,0 +1,56 @@ +//! PromQL conformance over a small corpus derived from **grafana/o11y-bench** +//! task-grading rubrics. +//! +//! Source: — 27 deduplicated `fact.query` +//! values (`backend: prometheus`) pulled from `tasks-spec/**/*.yaml` grading +//! rubrics (`tests/data/o11y_bench_promql.txt`). These are SRE-style +//! incident/investigation queries: error ratios, burn rate, cache-lag and +//! retry-backlog triage, capacity checks. +//! +//! LICENSE NOTE: o11y-bench is AGPL-3.0, unlike this repo's other vendored +//! corpora (MIT/Apache-2.0). Whether copying these short query fixtures +//! verbatim into a differently-licensed test suite is fine as-is is an open +//! question — tracked in issue #135, not resolved by this file's existence. +//! +//! We *lower* (parse → L2 → L3), we do not execute. Totality: every query +//! returns `Ok` or a clean `LoweringError` and never panics. Given the corpus +//! is small and hand-picked from realistic incident-response queries, we also +//! assert full lowering coverage — a regression here means a real pattern +//! broke, not statistical noise. + +use asap_frontend_promql::{lower_promql, PromqlError as LoweringError}; +use asap_ir::types::AccuracyTarget; + +const CORPUS: &str = include_str!("data/o11y_bench_promql.txt"); + +/// Non-comment, non-blank query lines. +fn queries() -> impl Iterator { + CORPUS + .lines() + .map(str::trim) + .filter(|l| !l.is_empty() && !l.starts_with('#')) +} + +#[test] +fn lowering_is_total_over_the_o11y_bench_corpus() { + let mut lowered = 0; + let mut rejected = Vec::new(); + + for q in queries() { + match lower_promql(q, AccuracyTarget::Exact) { + Ok(_) => lowered += 1, + Err(LoweringError::Parse(e)) => panic!("unexpected parse failure for {q:?}: {e}"), + Err(e) => rejected.push((q, e)), + } + } + + assert!( + rejected.is_empty(), + "expected every o11y-bench query to lower cleanly, but {} were rejected: {rejected:#?}", + rejected.len() + ); + assert_eq!( + lowered, 27, + "corpus size changed — update this assertion if the corpus was intentionally edited" + ); +}