Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions crates/frontend-promql/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -22,3 +22,7 @@ path = "tests/observability/awesome_prometheus_alerts.rs"
[[test]]
name = "promql_corpus"
path = "tests/observability/promql_corpus.rs"

[[test]]
name = "o11y_bench_promql"
path = "tests/observability/o11y_bench_promql.rs"
Original file line number Diff line number Diff line change
@@ -0,0 +1,37 @@
# PromQL query corpus derived from grafana/o11y-bench task-grading rubrics.
# Source: https://github.com/grafana/o11y-bench (tasks-spec/**/*.yaml, `fact.query`
# values where backend == prometheus). Snapshot fetched 2026-07-17; deduplicated.
# License: AGPL-3.0. Vendoring this corpus verbatim from an AGPL-3.0 source into a
# differently-licensed test suite has an open licensing question -- see issue #135.
# Do not treat this file's provenance as cleared; it is tracked, not resolved.
# One real-world PromQL expression per line, pulled from LLM-agent-benchmark grading
# rubrics (SRE-style incident/investigation queries: error ratios, burn rate, cache
# lag, retry backlog, capacity). Used by tests/observability/o11y_bench_promql.rs to
# pin parse + lowering behaviour over real-world queries (totality; no execution).
(sum by (job) (increase(http_requests_total{status=~"5..",job=~"user-service|order-service|payment-service"}[6h]))) / (sum by (job) (increase(http_requests_total{job=~"user-service|order-service|payment-service"}[6h])))
histogram_quantile(0.95, sum(rate(http_request_duration_seconds_bucket{job="order-service"}[5m])) by (le))
histogram_quantile(0.99, sum(rate(http_request_duration_seconds_bucket{job="order-service"}[6h])) by (le))
max((sum by (job) (rate(http_requests_total{status=~"5..",job=~"user-service|order-service|payment-service"}[6h]))) / (sum by (job) (rate(http_requests_total{job=~"user-service|order-service|payment-service"}[6h]))))
max_over_time((sum(rate(http_requests_total{job="order-service",status=~"5.."}[5m])) / sum(rate(http_requests_total{job="order-service"}[5m])))[6h:1m])
max_over_time(histogram_quantile(0.95, sum(rate(http_request_duration_seconds_bucket{job="user-service"}[5m])) by (le))[12h:1m])
max_over_time(service_cache_refresh_lag_seconds{job="user-service"}[12h])
max_over_time(service_retry_queue_depth{job="order-service"}[6h])
max_over_time(service_retry_queue_depth{job="payment-service"}[6h])
max_over_time(service_retry_queue_depth{job=~".+"}[6h])
max_over_time(sum(process_resident_memory_bytes)[6h:1m])
service_cache_refresh_lag_seconds{job="user-service"}
sort_desc(sum by (job) (rate(process_cpu_seconds_total{job=~".+"}[6h])))
sum by (job) (increase(http_requests_total{status=~"5..",job=~".+-service"}[24h])) > 0
sum by (job) (rate(process_cpu_seconds_total{job=~".+"}[6h]))
sum(avg_over_time(process_resident_memory_bytes{job=~".+"}[6h]))
sum(increase(http_requests_total{job="order-service",status=~"5.."}[24h])) / sum(increase(http_requests_total{status=~"5.."}[24h]))
sum(increase(http_requests_total{job="payment-service",status=~"5.."}[1h] offset 6h)) / sum(increase(http_requests_total{job="payment-service"}[1h] offset 6h))
sum(increase(http_requests_total{job="payment-service",status=~"5.."}[1h])) / sum(increase(http_requests_total{job="payment-service"}[1h]))
sum(process_resident_memory_bytes)
sum(rate(http_requests_total{job="order-service"}[5m] offset 1h))
sum(rate(http_requests_total{job="order-service"}[5m]))
sum(rate(http_requests_total{job=~"user-service|order-service|payment-service"}[5m]))
sum(rate(http_requests_total{status=~"5..",job=~"payment-service|order-service"}[6h])) / sum(rate(http_requests_total{job=~"payment-service|order-service"}[6h]))
sum(rate(http_requests_total{status=~"5..",job=~"user-service|order-service|payment-service"}[1h])) / sum(rate(http_requests_total{job=~"user-service|order-service|payment-service"}[1h]))
sum(rate(process_cpu_seconds_total[1h]))
sum(up)
56 changes: 56 additions & 0 deletions crates/frontend-promql/tests/observability/o11y_bench_promql.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,56 @@
//! PromQL conformance over a small corpus derived from **grafana/o11y-bench**
//! task-grading rubrics.
//!
//! Source: <https://github.com/grafana/o11y-bench> — 27 deduplicated `fact.query`
//! values (`backend: prometheus`) pulled from `tasks-spec/**/*.yaml` grading
//! rubrics (`tests/data/o11y_bench_promql.txt`). These are SRE-style
//! incident/investigation queries: error ratios, burn rate, cache-lag and
//! retry-backlog triage, capacity checks.
//!
//! LICENSE NOTE: o11y-bench is AGPL-3.0, unlike this repo's other vendored
//! corpora (MIT/Apache-2.0). Whether copying these short query fixtures
//! verbatim into a differently-licensed test suite is fine as-is is an open
//! question — tracked in issue #135, not resolved by this file's existence.
//!
//! We *lower* (parse → L2 → L3), we do not execute. Totality: every query
//! returns `Ok` or a clean `LoweringError` and never panics. Given the corpus
//! is small and hand-picked from realistic incident-response queries, we also
//! assert full lowering coverage — a regression here means a real pattern
//! broke, not statistical noise.

use asap_frontend_promql::{lower_promql, PromqlError as LoweringError};
use asap_ir::types::AccuracyTarget;

const CORPUS: &str = include_str!("data/o11y_bench_promql.txt");

/// Non-comment, non-blank query lines.
fn queries() -> impl Iterator<Item = &'static str> {
CORPUS
.lines()
.map(str::trim)
.filter(|l| !l.is_empty() && !l.starts_with('#'))
}

#[test]
fn lowering_is_total_over_the_o11y_bench_corpus() {
let mut lowered = 0;
let mut rejected = Vec::new();

for q in queries() {
match lower_promql(q, AccuracyTarget::Exact) {
Ok(_) => lowered += 1,
Err(LoweringError::Parse(e)) => panic!("unexpected parse failure for {q:?}: {e}"),
Err(e) => rejected.push((q, e)),
}
}

assert!(
rejected.is_empty(),
"expected every o11y-bench query to lower cleanly, but {} were rejected: {rejected:#?}",
rejected.len()
);
assert_eq!(
lowered, 27,
"corpus size changed — update this assertion if the corpus was intentionally edited"
);
}
Loading