diff --git a/Cargo.lock b/Cargo.lock index ab43962c3..93977c3ae 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -342,6 +342,7 @@ dependencies = [ name = "asap-frontend-metricsql" version = "0.1.0" dependencies = [ + "asap-frontend-common", "asap-types", "metricsql_parser", "thiserror 2.0.18", @@ -352,6 +353,7 @@ name = "asap-frontend-promql" version = "0.1.0" dependencies = [ "asap-aware-mapping", + "asap-frontend-common", "asap-types", "promql-parser", ] diff --git a/crates/frontend-metricsql/Cargo.toml b/crates/frontend-metricsql/Cargo.toml index 8fa341b80..cba166f7c 100644 --- a/crates/frontend-metricsql/Cargo.toml +++ b/crates/frontend-metricsql/Cargo.toml @@ -5,5 +5,6 @@ edition = "2021" [dependencies] asap-types = { path = "../types" } +asap-frontend-common = { path = "../frontend-common" } metricsql_parser = { path = "../metricsql-parser-vendored" } thiserror = "2" diff --git a/crates/frontend-metricsql/src/lib.rs b/crates/frontend-metricsql/src/lib.rs index 26a034c43..4417a0be3 100644 --- a/crates/frontend-metricsql/src/lib.rs +++ b/crates/frontend-metricsql/src/lib.rs @@ -324,3 +324,6 @@ fn require_arity(name: &str, actual: usize, expected: usize) -> Result<(), Metri fn unsupported(message: impl Into) -> MetricsqlError { MetricsqlError::UnsupportedFeature(message.into()) } + +/// Unified lowering, promoted to the root API at planner cutover. +pub mod unified; diff --git a/crates/frontend-metricsql/src/unified/mod.rs b/crates/frontend-metricsql/src/unified/mod.rs new file mode 100644 index 000000000..3544a9801 --- /dev/null +++ b/crates/frontend-metricsql/src/unified/mod.rs @@ -0,0 +1,389 @@ +//! MetricsQL AST → the name-based `UnresolvedOp` tree → the unified operator DAG. + +use std::{rc::Rc, time::Duration}; + +use asap_frontend_common::{ + resolve_root, UnresolvedOp as U, UnresolvedPredicate, UnresolvedScalar, +}; +use asap_types::ir::{BinaryOperator, ExprSemantics, OperatorNode, TimeRangeKind}; +use asap_types::pre_asap::{ + AggIntent, ArithmeticOpKind, BinaryOpKind, ColumnRef, CompareOpKind, GroupKeys, + PromQLVectorSetOpKind, Reduction, ScalarValue, Source, +}; +use asap_types::types::AccuracyTarget; +use metricsql_parser::ast::{AggregateModifier, DurationExpr, Expr, MetricExpr, RollupExpr}; +use metricsql_parser::functions::{AggregateFunction, BuiltinFunction, RollupFunction}; +use metricsql_parser::label::{LabelFilter, LabelFilterOp, NAME_LABEL}; +use thiserror::Error; + +pub use metricsql_parser::ast::Expr as MetricsqlExpr; + +#[derive(Debug, Error)] +pub enum MetricsqlError { + #[error("MetricsQL parse error: {0}")] + Parse(String), + #[error("unsupported MetricsQL feature: {0}")] + UnsupportedFeature(String), + #[error("MetricsQL column resolution failed: {0}")] + Resolve(String), +} + +pub fn parse_metricsql(query: &str) -> Result { + metricsql_parser::parser::parse(query).map_err(|e| MetricsqlError::Parse(e.to_string())) +} + +pub fn canonical_metricsql(query: &str) -> Result { + Ok(parse_metricsql(query)?.to_string()) +} + +pub fn lower_metricsql( + query: &str, + accuracy: AccuracyTarget, +) -> Result, MetricsqlError> { + match lower_metricsql_query(query, accuracy)? { + asap_types::ir::QueryRoot::Operator(node) => Ok(node), + _ => Err(unsupported("scalar root: use lower_metricsql_query")), + } +} + +/// Lower scalar constants without fabricating a relational operator. +pub fn lower_metricsql_query( + query: &str, + accuracy: AccuracyTarget, +) -> Result { + let ast = parse_metricsql(query)?; + if let Expr::NumberLiteral(number) = &ast { + return Ok(asap_types::ir::QueryRoot::Scalar( + asap_types::ir::ScalarExpr::literal_f64(number.value), + )); + } + let unresolved = Lowerer { accuracy }.lower(&ast)?; + resolve_root(&unresolved) + .map(asap_types::ir::QueryRoot::Operator) + .map_err(|e| MetricsqlError::Resolve(e.to_string())) +} + +struct Lowerer { + accuracy: AccuracyTarget, +} + +impl Lowerer { + fn lower(&self, expr: &Expr) -> Result { + match expr { + Expr::MetricExpression(e) => self.metric(e), + Expr::Rollup(e) => self.rollup(e), + Expr::Function(e) => self.function(e), + Expr::Aggregation(e) => self.aggregate(e), + Expr::NumberLiteral(_) => { + Err(unsupported("scalar root requires lower_metricsql_query")) + } + // Vector negation is `x * -1` (as in the PromQL front end). + Expr::UnaryOperator(e) => Ok(U::PromqlScalarOp { + child: Rc::new(self.lower(&e.expr)?), + scalar: UnresolvedScalar::Literal(ScalarValue::Float64(-1.0)), + op: BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), + scalar_left: false, + return_bool: false, + }), + Expr::BinaryOperator(e) => self.binary(e), + Expr::Parens(e) if e.expressions.len() == 1 => self.lower(&e.expressions[0]), + Expr::With(e) => self.lower(&e.expr), + other => Err(unsupported(format!("AST node `{other}`"))), + } + } + + fn metric(&self, metric: &MetricExpr) -> Result { + if metric.has_or_matchers() { + return Err(unsupported("or-delimited selector matchers")); + } + let name = metric + .metric_name() + .ok_or_else(|| unsupported("selector without one exact metric name"))?; + let mut filters: Vec<_> = metric + .matchers + .filter_iter() + .filter(|f| f.label != NAME_LABEL) + .collect(); + filters.sort_by(|a, b| a.label.cmp(&b.label).then(a.value.cmp(&b.value))); + Ok(U::Scan { + source: Source::TimeSeries { + metric: name.to_owned(), + }, + predicates: filters + .into_iter() + .map(|f| UnresolvedPredicate(matcher(f))) + .collect(), + schema: None, + }) + } + + fn rollup(&self, rollup: &RollupExpr) -> Result { + if rollup.offset.is_some() || rollup.at.is_some() { + return Err(unsupported("offset and @ modifiers")); + } + if rollup.for_subquery() { + return Err(unsupported("subquery step or inherited step")); + } + let child = self.lower(&rollup.expr)?; + match &rollup.window { + None => Ok(child), + Some(window) => Ok(U::TimeRange { + range: duration(window)?, + kind: TimeRangeKind::Range, + child: Rc::new(child), + }), + } + } + + fn function( + &self, + function: &metricsql_parser::ast::FunctionExpr, + ) -> Result { + if function.keep_metric_names { + return Err(unsupported( + "keep_metric_names requires metric-name lineage", + )); + } + let BuiltinFunction::Rollup(rollup) = function.function else { + return Err(unsupported(format!("function `{}`", function.name()))); + }; + let expected_args = if rollup == RollupFunction::QuantileOverTime { + 2 + } else { + 1 + }; + require_arity(function.name(), function.args.len(), expected_args)?; + let child_index = usize::from(rollup == RollupFunction::QuantileOverTime); + let child = function + .args + .get(child_index) + .ok_or_else(|| unsupported(format!("missing argument for `{}`", function.name())))?; + let intent = match rollup { + RollupFunction::DefaultRollup | RollupFunction::LastOverTime => AggIntent::LastOverTime, + RollupFunction::FirstOverTime => AggIntent::FirstOverTime, + RollupFunction::AvgOverTime => AggIntent::Avg { col: None }, + RollupFunction::MinOverTime => AggIntent::Min { col: None }, + RollupFunction::MaxOverTime => AggIntent::Max { col: None }, + RollupFunction::SumOverTime => AggIntent::Sum { col: None }, + RollupFunction::CountOverTime => AggIntent::Count { + accuracy: self.accuracy.clone(), + }, + RollupFunction::StddevOverTime => AggIntent::StdDev { + col: None, + population: true, + }, + RollupFunction::StdvarOverTime => AggIntent::Variance { + col: None, + population: true, + }, + RollupFunction::Rate => AggIntent::Rate, + RollupFunction::IRate => AggIntent::IRate, + RollupFunction::Increase => AggIntent::Increase, + RollupFunction::Changes => AggIntent::Changes, + RollupFunction::Delta => AggIntent::Delta, + RollupFunction::IDelta => AggIntent::IDelta, + RollupFunction::Deriv => AggIntent::Deriv, + RollupFunction::Resets => AggIntent::Resets, + RollupFunction::MadOverTime => AggIntent::MadOverTime, + RollupFunction::PresentOverTime => AggIntent::PresentOverTime, + RollupFunction::AbsentOverTime => AggIntent::AbsentOverTime, + RollupFunction::QuantileOverTime => AggIntent::Quantile { + col: None, + q: number_arg(&function.args, 0)?, + accuracy: self.accuracy.clone(), + }, + _ => { + return Err(unsupported(format!( + "rollup function `{}`", + function.name() + ))) + } + }; + let child = self.lower(child)?; + if rollup == RollupFunction::DefaultRollup && !matches!(child, U::TimeRange { .. }) { + return Err(unsupported( + "default_rollup without an explicit range requires an evaluation step", + )); + } + Ok(aggregate(Reduction::PerEntity, intent, child)) + } + + fn aggregate( + &self, + expr: &metricsql_parser::ast::AggregationExpr, + ) -> Result { + if expr.limit != 0 || expr.keep_metric_names { + return Err(unsupported("aggregate limit or keep_metric_names")); + } + let expected_args = if expr.function == AggregateFunction::Quantile { + 2 + } else { + 1 + }; + require_arity(expr.name(), expr.args.len(), expected_args)?; + let child_index = expr + .arg_idx_for_optimization() + .ok_or_else(|| unsupported(format!("aggregate `{}` arguments", expr.name())))?; + let intent = match expr.function { + AggregateFunction::Sum => AggIntent::Sum { col: None }, + AggregateFunction::Avg => AggIntent::Avg { col: None }, + AggregateFunction::Min => AggIntent::Min { col: None }, + AggregateFunction::Max => AggIntent::Max { col: None }, + AggregateFunction::Count => AggIntent::Cardinality { + cols: vec![], + accuracy: self.accuracy.clone(), + }, + AggregateFunction::StdDev => AggIntent::StdDev { + col: None, + population: true, + }, + AggregateFunction::StdVar => AggIntent::Variance { + col: None, + population: true, + }, + AggregateFunction::Group => AggIntent::Group, + AggregateFunction::Quantile => AggIntent::Quantile { + col: None, + q: number_arg(&expr.args, 0)?, + accuracy: self.accuracy.clone(), + }, + _ => return Err(unsupported(format!("aggregate `{}`", expr.name()))), + }; + let reduction = match &expr.modifier { + None => Reduction::by(vec![]), + Some(AggregateModifier::By(v)) => Reduction::by(names(v)), + Some(AggregateModifier::Without(v)) => Reduction::Reduce(GroupKeys::without(names(v))), + }; + let child = expr + .args + .get(child_index) + .ok_or_else(|| unsupported("missing aggregate input"))?; + Ok(aggregate(reduction, intent, self.lower(child)?)) + } + + fn binary(&self, expr: &metricsql_parser::ast::BinaryExpr) -> Result { + if expr.modifier.is_some() { + return Err(unsupported("binary vector matching modifiers")); + } + use metricsql_parser::ast::Operator as O; + let op = match expr.op { + O::Add => BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), + O::Sub => BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub), + O::Mul => BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul), + O::Div => BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), + O::Mod => BinaryOpKind::Arithmetic(ArithmeticOpKind::Mod), + O::Pow => BinaryOpKind::Arithmetic(ArithmeticOpKind::Pow), + O::Atan2 => BinaryOpKind::Arithmetic(ArithmeticOpKind::Atan2), + O::Eql => BinaryOpKind::Compare(CompareOpKind::Eq), + O::NotEq => BinaryOpKind::Compare(CompareOpKind::Ne), + O::Lt => BinaryOpKind::Compare(CompareOpKind::Lt), + O::Lte => BinaryOpKind::Compare(CompareOpKind::Le), + O::Gt => BinaryOpKind::Compare(CompareOpKind::Gt), + O::Gte => BinaryOpKind::Compare(CompareOpKind::Ge), + O::And => BinaryOpKind::Set(PromQLVectorSetOpKind::And), + O::Or => BinaryOpKind::Set(PromQLVectorSetOpKind::Or), + O::Unless => BinaryOpKind::Set(PromQLVectorSetOpKind::Unless), + O::If | O::IfNot | O::Default => { + return Err(unsupported(format!("MetricsQL operator `{}`", expr.op))) + } + }; + for (scalar, vector, scalar_left) in [ + (&expr.left, &expr.right, true), + (&expr.right, &expr.left, false), + ] { + if let Expr::NumberLiteral(n) = scalar.as_ref() { + return Ok(U::PromqlScalarOp { + child: Rc::new(self.lower(vector)?), + scalar: UnresolvedScalar::Literal(ScalarValue::Float64(n.value)), + op, + scalar_left, + return_bool: false, + }); + } + } + Ok(binary_op( + op, + self.lower(&expr.left)?, + self.lower(&expr.right)?, + )) + } +} + +/// A `BinaryOp` with default matching; MetricsQL modifiers (including `bool`) +/// are rejected before reaching here. +fn binary_op(kind: BinaryOpKind, lhs: U, rhs: U) -> U { + U::BinaryOp { + operator: BinaryOperator { + kind, + vector_match: None, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool: false, + lhs: Rc::new(lhs), + rhs: Rc::new(rhs), + } +} + +fn names(values: &[String]) -> Vec { + values.iter().cloned().map(ColumnRef::Named).collect() +} + +fn aggregate(reduction: Reduction, intent: AggIntent, child: U) -> U { + U::Aggregate { + reduction, + measures: vec![intent], + output_names: vec![String::new()], + filters: vec![], + having: None, + child: Rc::new(child), + } +} + +fn matcher(filter: &LabelFilter) -> UnresolvedScalar { + let op = match filter.op { + LabelFilterOp::Equal => CompareOpKind::Eq, + LabelFilterOp::NotEqual => CompareOpKind::Ne, + LabelFilterOp::RegexEqual => CompareOpKind::Regex, + LabelFilterOp::RegexNotEqual => CompareOpKind::NotRegex, + }; + UnresolvedScalar::Compare { + left: Box::new(UnresolvedScalar::Column(ColumnRef::Named( + filter.label.clone(), + ))), + op, + right: Box::new(UnresolvedScalar::Literal(ScalarValue::Utf8( + filter.value.clone(), + ))), + semantics: ExprSemantics::Promql, + } +} + +fn duration(value: &DurationExpr) -> Result { + match value { + DurationExpr::Millis(ms) if *ms >= 0 => Ok(Duration::from_millis(*ms as u64)), + DurationExpr::StepValue(_) => Err(unsupported("step-relative duration")), + DurationExpr::Millis(_) => Err(unsupported("negative duration")), + } +} + +fn number_arg(args: &[Expr], index: usize) -> Result { + match args.get(index) { + Some(Expr::NumberLiteral(v)) if v.value.is_finite() => Ok(v.value), + _ => Err(unsupported(format!("numeric argument #{index}"))), + } +} + +fn require_arity(name: &str, actual: usize, expected: usize) -> Result<(), MetricsqlError> { + if actual == expected { + Ok(()) + } else { + Err(unsupported(format!( + "`{name}` with {actual} arguments; canonical lowering requires exactly {expected}" + ))) + } +} + +fn unsupported(message: impl Into) -> MetricsqlError { + MetricsqlError::UnsupportedFeature(message.into()) +} diff --git a/crates/frontend-promql/Cargo.toml b/crates/frontend-promql/Cargo.toml index b108e32c7..b8576ae9b 100644 --- a/crates/frontend-promql/Cargo.toml +++ b/crates/frontend-promql/Cargo.toml @@ -7,6 +7,7 @@ edition = "2021" # — both in asap-types. Pulls the PromQL parser only — never DataFusion. [dependencies] asap-types = { path = "../types" } +asap-frontend-common = { path = "../frontend-common" } # Shared ProjectASAP parser contract. Keep this immutable revision aligned # with backend parsing and treat newly parsed functions as unsupported until diff --git a/crates/frontend-promql/src/lib.rs b/crates/frontend-promql/src/lib.rs index 31b169359..348fa0263 100644 --- a/crates/frontend-promql/src/lib.rs +++ b/crates/frontend-promql/src/lib.rs @@ -191,3 +191,6 @@ mod tests { )); } } + +/// Unified lowering, promoted to the root API at planner cutover. +pub mod unified; diff --git a/crates/frontend-promql/src/unified/error.rs b/crates/frontend-promql/src/unified/error.rs new file mode 100644 index 000000000..a889d6710 --- /dev/null +++ b/crates/frontend-promql/src/unified/error.rs @@ -0,0 +1,81 @@ +use std::fmt; + +use asap_frontend_common::ResolveDAGError; +use asap_types::workload::WorkloadError; + +/// Errors from lowering a PromQL query (parse → the name-based unresolved +/// tree, built directly → +/// [`resolve_root`](asap_frontend_common::resolve_root) binds it to the +/// unified operator DAG, issue #179). +/// +/// Carries no DataFusion type — the PromQL front end never depends on the SQL +/// stack. The language-neutral variants (`UnsupportedFeature` / `WrongLanguage` +/// / `Convert`) are mirrored by [`asap_frontend_sql::SqlError`] rather than +/// shared, so neither front end pulls the other's parser. +#[derive(Debug)] +pub enum PromqlError { + /// The workload omitted information required for plan-ready PromQL lowering. + InvalidWorkload(WorkloadError), + /// The `promql-parser` crate rejected the query string (parse failure). + Parse(String), + /// A PromQL function (`rate`, `*_over_time`, …) not supported in this version. + UnsupportedFunction(String), + /// A PromQL aggregation operator (`sum`, `topk`, …) not supported. + UnsupportedAggregateOp(String), + /// A structural feature (offset / `@` / `without`) not supported in this + /// version. + UnsupportedFeature(String), + /// A required function / aggregator argument was missing. + MissingArgument(String), + /// An argument had the wrong shape (e.g. a non-numeric `topk` parameter). + InvalidParameter(String), + /// The workload's query language is not PromQL. + WrongLanguage(String), + /// Resolving the canonical unresolved tree failed (name resolution + /// against the bound schema). + Convert(ResolveDAGError), +} + +impl fmt::Display for PromqlError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::InvalidWorkload(e) => write!(f, "invalid PromQL workload: {e}"), + Self::Parse(e) => write!(f, "PromQL parse error: {e}"), + Self::UnsupportedFunction(n) => write!(f, "unsupported PromQL function: {n}"), + Self::UnsupportedAggregateOp(n) => write!(f, "unsupported PromQL aggregate op: {n}"), + Self::UnsupportedFeature(m) => write!(f, "unsupported feature: {m}"), + Self::MissingArgument(m) => write!(f, "missing argument: {m}"), + Self::InvalidParameter(m) => write!(f, "invalid parameter: {m}"), + Self::WrongLanguage(l) => write!(f, "unsupported query language: {l}"), + Self::Convert(e) => write!(f, "column resolution failed: {e}"), + } + } +} + +impl std::error::Error for PromqlError {} + +impl From for PromqlError { + fn from(e: ResolveDAGError) -> Self { + Self::Convert(e) + } +} + +impl From for PromqlError { + fn from(e: WorkloadError) -> Self { + Self::InvalidWorkload(e) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn unsupported_feature_label_is_language_neutral() { + // `UnsupportedFeature` shares a Display label with the SQL side, so it + // must not hardcode "PromQL". + let msg = PromqlError::UnsupportedFeature("subquery".into()).to_string(); + assert_eq!(msg, "unsupported feature: subquery"); + assert!(!msg.contains("PromQL"), "got: {msg}"); + } +} diff --git a/crates/frontend-promql/src/unified/histogram.rs b/crates/frontend-promql/src/unified/histogram.rs new file mode 100644 index 000000000..ecb8cd2c4 --- /dev/null +++ b/crates/frontend-promql/src/unified/histogram.rs @@ -0,0 +1,129 @@ +//! Sample-type metadata for the `histogram_quantile` discrimination (issue #79). +//! +//! Classic cumulative buckets use exact interpolation. The explicitly declared +//! `RawSamples` extension permits generic quantile sketches; it is not standard +//! PromQL histogram semantics. Native samples are rejected until the IR has a +//! native histogram sample type. Undeclared metrics require classic bucket +//! evidence (`by (le)`, a `_bucket` metric, or an `le` matcher). + +use std::cell::RefCell; +use std::collections::HashMap; + +/// The physical sample type behind a histogram metric — the true signal for +/// whether `histogram_quantile` over it can be re-sketched. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum HistogramKind { + /// Classic cumulative `le` buckets — pre-aggregated counts. The + /// distribution can't be reconstructed from them, so it is **not** + /// sketch-able: `histogram_quantile` is exact bucket interpolation. + ClassicBucket, + /// Native histogram samples; currently rejected because the IR lacks their type. + Native, + /// Raw float samples the client retains — sketch-able. This is the case the + /// generic `Quantile` lowering exists for (a client holding raw samples can + /// build a quantile sketch even though the user wrote `histogram_quantile`). + RawSamples, +} + +impl HistogramKind { + /// Whether `histogram_quantile` over this kind lowers to the sketch-able + /// generic `Quantile` (`true`) rather than exact bucket interpolation. + pub fn is_sketchable(self) -> bool { + matches!(self, HistogramKind::RawSamples) + } +} + +/// Metric-name → declared [`HistogramKind`]. Supplied by a client that knows its +/// sample types, to drive the `histogram_quantile` discrimination from metadata +/// instead of query structure (issue #79). +#[derive(Debug, Clone, Default)] +pub struct HistogramCatalog(HashMap); + +impl HistogramCatalog { + pub fn new() -> Self { + Self::default() + } + + /// Declare `metric`'s sample type (builder style). + pub fn with(mut self, metric: impl Into, kind: HistogramKind) -> Self { + self.0.insert(metric.into(), kind); + self + } + + /// The declared kind for `metric`, if any. + pub fn kind_of(&self, metric: &str) -> Option { + self.0.get(metric).copied() + } + + pub fn is_empty(&self) -> bool { + self.0.is_empty() + } +} + +thread_local! { + static CURRENT: RefCell> = const { RefCell::new(None) }; +} + +/// RAII guard installing `catalog` as the ambient histogram catalog for the +/// current thread, restoring the prior value on drop. +/// +/// Lowering is synchronous and processes one query at a time, so a thread-local +/// ambient catalog cleanly injects this read-only metadata into the deep, +/// free-function `walk` recursion without threading a parameter through every +/// signature (the discrimination is consulted in exactly one place, +/// `walk_histogram`). +pub(crate) struct CatalogGuard(Option); + +impl CatalogGuard { + pub(crate) fn install(catalog: HistogramCatalog) -> Self { + let prev = CURRENT.with(|c| c.borrow_mut().replace(catalog)); + CatalogGuard(prev) + } +} + +impl Drop for CatalogGuard { + fn drop(&mut self) { + CURRENT.with(|c| *c.borrow_mut() = self.0.take()); + } +} + +/// The ambient catalog's declared kind for `metric`, or `None` when no catalog +/// is installed or the metric is undeclared (→ fall back to the heuristic). +pub(crate) fn current_kind_of(metric: &str) -> Option { + CURRENT.with(|c| c.borrow().as_ref().and_then(|cat| cat.kind_of(metric))) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn only_explicit_raw_samples_are_sketchable() { + assert!(!HistogramKind::ClassicBucket.is_sketchable()); + assert!(!HistogramKind::Native.is_sketchable()); + assert!(HistogramKind::RawSamples.is_sketchable()); + } + + #[test] + fn catalog_lookup() { + let cat = HistogramCatalog::new() + .with("classic", HistogramKind::ClassicBucket) + .with("raw", HistogramKind::RawSamples); + assert_eq!(cat.kind_of("classic"), Some(HistogramKind::ClassicBucket)); + assert_eq!(cat.kind_of("raw"), Some(HistogramKind::RawSamples)); + assert_eq!(cat.kind_of("unknown"), None); + } + + #[test] + fn guard_installs_and_restores_the_ambient_catalog() { + assert_eq!(current_kind_of("m"), None); + { + let _g = CatalogGuard::install( + HistogramCatalog::new().with("m", HistogramKind::ClassicBucket), + ); + assert_eq!(current_kind_of("m"), Some(HistogramKind::ClassicBucket)); + } + // Restored to empty after the guard drops. + assert_eq!(current_kind_of("m"), None); + } +} diff --git a/crates/frontend-promql/src/unified/mod.rs b/crates/frontend-promql/src/unified/mod.rs new file mode 100644 index 000000000..e7d99fe4c --- /dev/null +++ b/crates/frontend-promql/src/unified/mod.rs @@ -0,0 +1,233 @@ +//! PromQL front end: parse (via `promql-parser`) → the name-based +//! [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) tree, built directly +//! in canonical shape (issue #179) → [`resolve_root`]. +//! +//! `resolve_root` runs the +//! [`SchemaResolver`](asap_frontend_common::SchemaResolver) for positional +//! name resolution and returns the unified +//! [`OperatorNode`](asap_types::ir::OperatorNode) DAG. Depends on the PromQL +//! parser only — never on the SQL / DataFusion stack. + +pub mod error; +pub mod histogram; +pub mod promql; + +use std::rc::Rc; + +use asap_types::ir::OperatorNode; +use asap_types::workload::{DurationMs, PlanningWorkload, QueryLanguage, WorkloadError}; + +pub use error::PromqlError; +pub use histogram::{HistogramCatalog, HistogramKind}; + +/// Lower every normalized PromQL workload entry to a plan-ready operator DAG. +/// +/// PromQL workloads must declare a non-zero `data_ingestion_interval`; it is +/// injected around each bare instant selector. Explicit range selectors keep +/// their query-specified range. +/// `now_ms` is the planning time in Unix milliseconds; cadence evidence must +/// be valid at that time, using the same clock as downstream planning. +pub fn lower_promql_workload( + workload: &PlanningWorkload, + now_ms: u64, +) -> Result>, PromqlError> { + lower_promql_workload_inner(workload, now_ms) +} + +/// Like [`lower_promql_workload`], but uses `histograms` to distinguish classic +/// bucket interpolation from generic sketchable quantiles. +pub fn lower_promql_workload_with_histograms( + workload: &PlanningWorkload, + histograms: HistogramCatalog, + now_ms: u64, +) -> Result>, PromqlError> { + let _guard = histogram::CatalogGuard::install(histograms); + lower_promql_workload_inner(workload, now_ms) +} + +/// Lower scalar and vector query roots without introducing constant operators. +pub fn lower_promql_query_workload( + workload: &PlanningWorkload, + now_ms: u64, +) -> Result, PromqlError> { + lower_promql_query_workload_inner(workload, now_ms) +} + +pub fn lower_promql_query_workload_with_histograms( + workload: &PlanningWorkload, + histograms: HistogramCatalog, + now_ms: u64, +) -> Result, PromqlError> { + let _guard = histogram::CatalogGuard::install(histograms); + lower_promql_query_workload_inner(workload, now_ms) +} + +fn lower_promql_workload_inner( + workload: &PlanningWorkload, + now_ms: u64, +) -> Result>, PromqlError> { + lower_promql_query_workload_inner(workload, now_ms)? + .into_iter() + .map(|root| match root { + asap_types::ir::QueryRoot::Operator(node) => Ok(node), + asap_types::ir::QueryRoot::Scalar(_) => Err(PromqlError::UnsupportedFeature( + "scalar root: use lower_promql_query_workload".into(), + )), + }) + .collect() +} + +fn lower_promql_query_workload_inner( + workload: &PlanningWorkload, + now_ms: u64, +) -> Result, PromqlError> { + if !matches!(workload.query_workload.language, QueryLanguage::PromQL) { + return Err(PromqlError::WrongLanguage(format!( + "{:?}", + workload.query_workload.language + ))); + } + workload.validate()?; + let &DurationMs(interval_ms) = workload + .data_workload + .as_ref() + .expect("validated PromQL workload has data_workload") + .data_ingestion_interval + .value_at(now_ms) + .ok_or(WorkloadError::UnavailableDataIngestionInterval)?; + workload + .query_workload + .entries() + .map(|entry| { + let root = promql::PromqlLowerer::lower_query_with_ingestion_interval( + &entry.query.0, + &entry.requirements.accuracy.target(), + std::time::Duration::from_millis(interval_ms), + )?; + Ok(root) + }) + .collect() +} + +#[cfg(test)] +mod tests { + // Expiring evidence without an observation timestamp is never usable. + #[test] + fn rejects_unusable_ingestion_evidence() { + let mut input = workload("sum(data)"); + input + .data_workload + .as_mut() + .unwrap() + .data_ingestion_interval + .valid_for_ms = Some(100); + assert!(lower_promql_workload(&input, 0).is_err()); + } + + // Cadence expiry is inclusive; future and expired evidence cannot set a horizon. + #[test] + fn ingestion_evidence_respects_planning_time_with_and_without_histograms() { + let mut input = workload("sum(data)"); + let evidence = &mut input + .data_workload + .as_mut() + .unwrap() + .data_ingestion_interval; + evidence.observed_at_ms = Some(1_000); + evidence.valid_for_ms = Some(100); + for (now_ms, usable) in [(999, false), (1_000, true), (1_100, true), (1_101, false)] { + assert_eq!(lower_promql_workload(&input, now_ms).is_ok(), usable); + assert_eq!( + lower_promql_workload_with_histograms(&input, HistogramCatalog::default(), now_ms) + .is_ok(), + usable + ); + } + input + .data_workload + .as_mut() + .unwrap() + .data_ingestion_interval + .observed_at_ms = None; + assert!( + lower_promql_workload_with_histograms(&input, HistogramCatalog::default(), 1_000) + .is_err() + ); + } + use std::time::Duration; + + use asap_types::ir::{NonASAPOp, TimeRangeKind}; + use asap_types::workload::{ + BatchEntry, DataWorkload, Evidence, PlanningWorkload, Query, QueryRequirements, + QueryWorkload, TimeSelection, + }; + + use super::*; + + fn workload(query: &str) -> PlanningWorkload { + PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements::default(), + predictability: Default::default(), + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(1_000)), + ..Default::default() + }, + ..Default::default() + }), + } + } + + // A bare instant selector reads the latest sample within the declared + // ingestion interval: an `Instant` lookback of that length. + #[test] + fn instant_selector_uses_declared_ingestion_interval() { + let query = lower_promql_workload(&workload("sum by (job) (data)"), 0).unwrap(); + let NonASAPOp::Aggregate { child, .. } = query[0].expect_non_asap() else { + panic!("expected aggregate") + }; + assert!( + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, kind, child } + if *range == Duration::from_secs(1) + && *kind == TimeRangeKind::Instant + && matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) + ); + } + + // An explicit `m[5m]` keeps its own window as a `Range` selection. + #[test] + fn explicit_range_selector_keeps_its_query_range() { + let query = lower_promql_workload(&workload("sum_over_time(data[5m])"), 0).unwrap(); + let NonASAPOp::Aggregate { child, .. } = query[0].expect_non_asap() else { + panic!("expected aggregate") + }; + assert!( + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, kind, child } + if *range == Duration::from_secs(300) + && *kind == TimeRangeKind::Range + && matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) + ); + } + + #[test] + fn workload_without_interval_fails_loudly() { + let mut workload = workload("sum(data)"); + workload.data_workload = Some(DataWorkload::default()); + assert!(matches!( + lower_promql_workload(&workload, 0), + Err(PromqlError::InvalidWorkload( + asap_types::workload::WorkloadError::MissingDataIngestionInterval + )) + )); + } +} diff --git a/crates/frontend-promql/src/unified/promql.rs b/crates/frontend-promql/src/unified/promql.rs new file mode 100644 index 000000000..f97e4fb9d --- /dev/null +++ b/crates/frontend-promql/src/unified/promql.rs @@ -0,0 +1,2236 @@ +//! PromQL string → the name-based +//! [`UnresolvedOp`](asap_frontend_common::UnresolvedOp) tree. +//! +//! - **Parsing** is delegated to `promql-parser` 0.8. +//! - **Lowering** builds *directly in canonical shape* here (issue #179): the +//! walk interprets PromQL semantics (range vectors, aggregate operators, +//! label matchers) and emits `UnresolvedOp` / `UnresolvedScalar` nodes with +//! unresolved `ColumnRef`s — the same tree shape +//! [`resolve_root`](asap_frontend_common::resolve_root) later binds to the +//! positional [`OperatorNode`](asap_types::ir::OperatorNode) DAG. The structural decisions a +//! separate converter stage would otherwise have to make (heavy-hitter +//! `topk` recognition, the `PerEntity`/`Reduce` reduction choice, +//! `without(...)` grouping) are made right here, since a front end +//! building this shape already knows the answer at parse time — see +//! `reduction_for` and `mark_without`. `resolve_root` is left with exactly +//! the schema-*dependent* work: binding every `ColumnRef` to its +//! positional `ColumnId`. +//! +//! # PromQL → canonical unresolved-tree mapping (summary) +//! +//! | PromQL | Canonical shape | +//! |---|---| +//! | `quantile_over_time(φ, m{f}[w])` | `Aggregate{[Quantile(φ)], TimeRange{w, Scan{predicates}}}` | +//! | `histogram_quantile(φ, )` | `Aggregate{without(le), [HistogramQuantile(φ, le)]}` — cumulative-bucket interpolation (classic form recognised by `by (le)` / a `_bucket` metric / an `le` matcher) | +//! | `histogram_quantile(φ, )` | `Aggregate{[Quantile(φ)]}` over the fully-lowered arg (generic, sketch-able with an accuracy target) | +//! | `histogram_quantiles(v, "l", φ…)` | `Concat{PromqlRelabel{l=φᵢ, }…}` — one branch per φ (issue #109) | +//! | `histogram_count/sum/avg/stddev/stdvar(v)`, `histogram_fraction(l,u,v)` | `Aggregate{[Histogram*]}` — per-series native-histogram accessors (issue #43) | +//! | `OUTER_op(inner_func(m[w]))` (e.g. `sum(rate(m[w]))`) | `Aggregate{[OUTER_op]}` over `Aggregate{[inner_func]}` — two levels | +//! | `OUTER_op()` (e.g. `max(sum by (job) (rate(m[w])))`, `sum(rate(a[w]) + rate(b[w]))`) | `Aggregate{[OUTER_op]}` over the fully-lowered `` — arbitrary function nesting (issue #27) | +//! | `topk(k, )` / `bottomk(k, )` | `Sort{value} → Limit{k}` over the fully-lowered argument | +//! | `avg/min/max/sum_over_time(m[w])` | `Aggregate{[Avg/Min/Max/Sum], TimeRange{w}}` | +//! | `stddev/stdvar_over_time(m[w])` | `Aggregate{[StdDev/Variance], TimeRange{w}}` | +//! | `count_over_time(m[w])` | `Aggregate{[Count], TimeRange{w}}` | +//! | `last/first/mad/ts_of_min/ts_of_max/ts_of_first/ts_of_last_over_time(m[w])` | `Aggregate{[Last/First/Mad/TsOf…OverTime], TimeRange{w}}` — per-series range reducers (issue #51) | +//! | `sort`/`sort_desc(v)`, `sort_by_label[_desc](v,"l"…)` | `Sort{value \| label…}` (no `Limit`) — row-preserving reorder (issue #51); `min_of`/`max_of` scalar reducers → #89 | +//! | `rate(m[w])` / `irate(m[w])` | `Aggregate{[Rate/IRate], TimeRange{w}}` — distinct function identities; shared physical machinery is a later realization choice | +//! | `increase(m[w])` | `Aggregate{[Increase], TimeRange{w}}` | +//! | `changes`/`delta`/`idelta`/`deriv`/`resets`/`predict_linear`/`double_exponential_smoothing`(`m[w]`, …) | `Aggregate{[Changes/Delta/…], TimeRange{w}}` — per-series counter-derivative intents (issue #44) | +//! | `absent(v)` / `absent_over_time(m[w])` / `present_over_time(m[w])` | `Aggregate{[Absent/AbsentOverTime/PresentOverTime]}` — presence intents; the empty→synthesized-sample logic is a post-ASAP concern (issue #47) | +//! | `abs`/`ceil`/`sqrt`/`ln`/`clamp*`/`round`/trig(`v`), `pi()` | typed scalar `Project` (issue #45); `pi()` → a `ScalarExpr::Literal` root | +//! | `time()` / `timestamp`/`hour`/`day_of_week`/… (`v`) | `ScalarExpr::EvalTimestamp` root / `Aggregate{[TimeFn(f)]}` (issue #46) | +//! | `vector(s)` / `scalar(v)` | `PromqlVectorFromScalar(s)` / `ScalarExpr::PromqlScalarFromVector(v)` — the scalar⇄vector bridges (issue #48) | +//! | ` op ` (`time() - 1`, `1 < bool 2`, `-time()`) | `ScalarExpr::{Arithmetic, Case, Negative}` — a scalar expression, never an operator | +//! | `v op `, `a op bool b`, `v > bool 0` | `Project`/`Filter` with owned scalar expressions; vector/vector uses `BinaryOp{return_bool}` | +//! | `label_replace(v,…)` / `label_join(v,…)` | `PromqlRelabel{dst, value}` — per-series label rewrite; value unchanged (issue #50) | +//! | `info(v, [selector])` | `PromqlInfoEnrich{selector}` — label-enrichment join against the info metric(s); join keys resolved during post-ASAP binding (issue #84) | +//! | `group` / `offset` / `@` / `info` | **rejected** — distinct semantics with no intent-algebra representation yet (`info` label-join → #84) | +//! | `OUTER by (dims) (…)` | `Aggregate.reduction = Reduce(by = dims)` (generic `topk by`/`bottomk` grouping → `Sort.partition_by`) | +//! | `count by (d) (…)` | `Aggregate{[Count], …}` | +//! | `group(v)` / `count_values("l", v)` | `Aggregate{[Group]}` (constant 1) / `Aggregate{[CountValues{l}]}` (group-by-value + count, new label `l`) — issue #49 | +//! | `limitk(k, v)` / `limit_ratio(r, v)` | `PromqlSeriesSample{LimitK(k) \| LimitRatio(r)}` — series-sampling selection, whole series kept unchanged (issue #86) | +//! | `topk(k, count_over_time(…))` / `topk(k, sum_over_time(…))` | `Aggregate{[TopK{k}]}` (heavy-hitter intent) over the explicit inner `Aggregate{[Count/Sum]}` | +//! | `topk(k, )` / `bottomk(k, …)` | `Sort{value} → Limit{k}` | +//! | `m{f}` / `m{f}[w]` | `TimeRange{ingestion, Instant, Scan{predicates}}` / `TimeRange{w, Range, Scan}` | +//! | `a OP b` | `BinaryOp{vector_match}` | +//! | `expr[r:res]` | `PromqlSubquery{r, res}` | +//! | ` offset ` / ` @ `/`start()`/`end()` | `TimeShift{shift}` over the selector's `Scan` — pass-through schema; a ranged selector shifts under its `TimeRange` (issue #40) | + +use std::rc::Rc; +use std::time::{Duration, SystemTime}; + +use promql_parser::label::{MatchOp, Matcher}; +use promql_parser::parser::value::ValueType; +use promql_parser::parser::{ + self, token, AggregateExpr, AtModifier as ParserAtModifier, BinaryExpr, Call, Expr, + LabelModifier, Offset, VectorMatchCardinality, VectorSelector, +}; + +use asap_frontend_common::{ + UnresolvedOp as Unresolved, UnresolvedPredicate, UnresolvedScalar as Scalar, UnresolvedSortKey, +}; +use asap_types::ir::operator_properties::{ + AtModifier, BinaryOpKind, GroupKeys, GroupSide, PromQLVectorSetOpKind, Reduction, Source, + TimeShift, VectorGrouping, VectorMatch, VectorMatchKind, +}; +use asap_types::ir::{BinaryOperator, ExprSemantics, TimeRangeKind}; +use asap_types::pre_asap::agg_intent::{topk, AggIntent, TimeFunc}; + +use asap_types::pre_asap::{ + ArithmeticOpKind, ColumnRef, CompareOpKind, InfoMatcher, SampleKind, ScalarValue, +}; +use asap_types::types::AccuracyTarget; + +/// Every scalar expression this front end builds follows PromQL's numeric rules. +const PROMQL: ExprSemantics = ExprSemantics::Promql; + +use crate::unified::error::PromqlError as LoweringError; + +type Result = std::result::Result; + +/// Parses and lowers (→ the canonical, unresolved tree) a PromQL query string. +pub(crate) struct PromqlLowerer; + +#[derive(Debug, Clone)] +enum Outer { + None, + Plain(OuterIntent), + Count, + /// `count_values("l", v)` — group by value + count, emitting the value as a + /// new label `l` (issue #49). + CountValues { + label: String, + }, + TopK { + k: u64, + descending: bool, + }, + /// `limitk`/`limit_ratio` — series-sampling selection (issue #86). + Sample { + kind: SampleKind, + }, +} + +#[derive(Debug, Clone)] +enum OuterIntent { + Sum, + Avg, + Min, + Max, + StdDev, + Variance, + Quantile(f64), + /// `group(v)` — constant 1 per group (issue #49). + Group, +} + +#[derive(Debug, Clone)] +enum InnerFunc { + FrequencyL2, + FrequencyEntropy, + Cardinality, + Quantile(f64), + Avg, + Min, + Max, + Sum, + StdDev, + Variance, + Count, + // `Rate`/`Increase` carry no window of their own — unlike the old Unresolved + // `AggFunc::Rate{window}`, canonical `AggIntent::Rate`/`Increase` have no + // window field either; `windowed_aggregate` reads `Inner.window` + // uniformly for every intent, so it would be a redundant duplicate here. + Rate, + IRate, + Increase, + // Counter-derivative range functions (issue #44). The window rides on the + // enclosing `TimeRange` node (like `*_over_time`), so these carry only + // their non-window scalar params. + Changes, + Delta, + IDelta, + Deriv, + Resets, + PredictLinear(f64), + DoubleExp { smoothing: f64, trend: f64 }, + // Additional range-vector reducers (issue #51). Per-series over the window + // (like `*_over_time`); the window rides on the enclosing Unresolved `Window`. + LastOverTime, + FirstOverTime, + MadOverTime, + TsOfMinOverTime, + TsOfMaxOverTime, + TsOfFirstOverTime, + TsOfLastOverTime, +} + +struct Inner { + metric: String, + matchers: Vec, + window: Option, + func: Option, + /// `offset` / `@` on the selector, carried to the `Source` (issue #40). + shift: TimeShift, +} + +/// Maximum PromQL expression nesting depth the walker accepts. Real queries +/// nest only a handful deep; this bounds the recursive descent (`walk` and the +/// mutually-recursive helpers) so a pathologically nested query is rejected +/// rather than overflowing the stack. +const MAX_DEPTH: usize = 256; + +impl PromqlLowerer { + pub(crate) fn lower_query_with_ingestion_interval( + query: &str, + accuracy: &AccuracyTarget, + interval: Duration, + ) -> Result { + let _guard = AccuracyGuard::install(accuracy.clone()); + let _interval = IngestionIntervalGuard::install(interval); + let ast = parser::parse(query).map_err(LoweringError::Parse)?; + check_depth(&ast, MAX_DEPTH)?; + let mut metrics = Vec::new(); + collect_metric_names(&ast, &mut metrics); + if metrics.iter().any(|metric| { + crate::unified::histogram::current_kind_of(metric) + == Some(crate::unified::histogram::HistogramKind::Native) + }) { + return Err(LoweringError::UnsupportedFeature( + "native histogram samples have no IR representation".into(), + )); + } + + if ast.value_type() == ValueType::Scalar { + Ok(asap_types::ir::QueryRoot::Scalar( + asap_frontend_common::resolve_scalar_root(&lower_scalar(&ast)?)?, + )) + } else { + Ok(asap_types::ir::QueryRoot::Operator( + asap_frontend_common::resolve_root(&walk(&ast)?)?, + )) + } + } +} + +std::thread_local! { + static ACCURACY: std::cell::RefCell = + const { std::cell::RefCell::new(AccuracyTarget::Exact) }; + static INGESTION_INTERVAL: std::cell::RefCell> = const { std::cell::RefCell::new(None) }; +} + +/// RAII guard installing `accuracy` as the ambient accuracy target for the +/// current thread's lowering, restoring the prior value on drop — same shape +/// as `histogram::CatalogGuard`. +struct AccuracyGuard(AccuracyTarget); + +impl AccuracyGuard { + fn install(accuracy: AccuracyTarget) -> Self { + let prev = ACCURACY.with(|a| a.replace(accuracy)); + AccuracyGuard(prev) + } +} + +impl Drop for AccuracyGuard { + fn drop(&mut self) { + ACCURACY.with(|a| *a.borrow_mut() = std::mem::replace(&mut self.0, AccuracyTarget::Exact)); + } +} + +/// The ambient accuracy target installed by the current [`PromqlLowerer::lower`] call. +fn current_accuracy() -> AccuracyTarget { + ACCURACY.with(|a| a.borrow().clone()) +} + +struct IngestionIntervalGuard(Option); + +impl IngestionIntervalGuard { + fn install(interval: Duration) -> Self { + Self(INGESTION_INTERVAL.with(|current| current.replace(Some(interval)))) + } +} + +impl Drop for IngestionIntervalGuard { + fn drop(&mut self) { + INGESTION_INTERVAL.with(|current| *current.borrow_mut() = self.0.take()); + } +} + +fn current_ingestion_interval() -> Duration { + INGESTION_INTERVAL.with(|current| { + current + .borrow() + .expect("ingestion interval is installed for workload lowering") + }) +} + +/// Bounded depth check over the parser AST: errors once nesting would exceed +/// `budget` frames, descending into every child expression. +fn check_depth(expr: &Expr, budget: usize) -> Result<()> { + let Some(budget) = budget.checked_sub(1) else { + return Err(LoweringError::UnsupportedFeature(format!( + "query nesting exceeds the {MAX_DEPTH}-level limit" + ))); + }; + match expr { + Expr::Aggregate(a) => { + check_depth(&a.expr, budget)?; + if let Some(p) = &a.param { + check_depth(p, budget)?; + } + } + Expr::Unary(u) => check_depth(&u.expr, budget)?, + Expr::Binary(b) => { + check_depth(&b.lhs, budget)?; + check_depth(&b.rhs, budget)?; + } + Expr::Paren(p) => check_depth(&p.expr, budget)?, + Expr::Subquery(s) => check_depth(&s.expr, budget)?, + Expr::Call(c) => { + for arg in &c.args.args { + check_depth(arg, budget)?; + } + } + Expr::MatrixSelector(_) + | Expr::VectorSelector(_) + | Expr::NumberLiteral(_) + | Expr::StringLiteral(_) + | Expr::Extension(_) => {} + } + Ok(()) +} + +fn walk(expr: &Expr) -> Result { + // A scalar-typed expression (`5`, `time() - 1`, `scalar(v)`, `1 < bool 2`) + // is a scalar expression at an operator position, never an operator tree. + if expr.value_type() == ValueType::Scalar { + return Err(LoweringError::UnsupportedFeature( + "scalar root requires query-root lowering".into(), + )); + } + match expr { + Expr::Aggregate(agg) => walk_aggregate(agg), + Expr::Call(call) if call.func.name.starts_with("histogram_") => walk_histogram(call), + Expr::Call(call) if is_math_fn(call.func.name) => walk_math(call), + Expr::Call(call) if is_presence_fn(call.func.name) => walk_presence(call), + Expr::Call(call) if is_time_fn(call.func.name) => walk_time(call), + Expr::Call(call) if is_typeconv_fn(call.func.name) => walk_typeconv(call), + Expr::Call(call) if is_label_fn(call.func.name) => walk_label(call), + Expr::Call(call) if is_sort_fn(call.func.name) => walk_sort(call), + Expr::Call(call) if call.func.name == "info" => walk_info(call), + Expr::Call(call) => walk_call(call), + Expr::Binary(bin) => walk_binary(bin), + Expr::Paren(p) => walk(&p.expr), + // `UnaryExpr` is built only by negation (`Neg`); unary `+` is folded to + // identity and `-` to a negated `NumberLiteral`. A scalar + // operand was dispatched to `lower_scalar` above (→ `Negative`), so this + // is a vector projection. Unary negation retains the metric name. + Expr::Unary(u) => Ok(Unresolved::PromqlMap { + child: Rc::new(walk(&u.expr)?), + sample: Scalar::Negative { + expr: Box::new(Scalar::Column(ColumnRef::SampleValue)), + semantics: ExprSemantics::Promql, + }, + drop_metric_name: false, + }), + Expr::Subquery(sq) => { + let subquery = Unresolved::PromqlSubquery { + range: sq.range, + resolution: sq.step, + child: Rc::new(walk(&sq.expr)?), + }; + // `offset`/`@` move the whole subquery, including its step grid. + let shift = time_shift(sq.offset.as_ref(), sq.at.as_ref())?; + Ok(if shift.is_identity() { + subquery + } else { + Unresolved::TimeShift { + shift, + child: Rc::new(subquery), + } + }) + } + Expr::VectorSelector(vs) => { + let (metric, matchers, shift) = vs_parts(vs)?; + Ok(instant_source(metric, matchers, shift)) + } + Expr::MatrixSelector(ms) => { + let (metric, matchers, shift) = vs_parts(&ms.vs)?; + Ok(Unresolved::TimeRange { + range: ms.range, + kind: TimeRangeKind::Range, + child: Rc::new(filtered_source(metric, matchers, shift)), + }) + } + // Scalar-typed, dispatched above; kept for exhaustiveness. String + // literals only appear as function args (`label_replace`, …), so a + // bare one is rejected (issue #35). + Expr::NumberLiteral(_) => unreachable!("scalar handled above"), + Expr::StringLiteral(_) => Err(LoweringError::UnsupportedFeature( + "bare string literal".into(), + )), + Expr::Extension(_) => Err(LoweringError::UnsupportedFeature( + "extension expression".into(), + )), + } +} + +/// Lower a scalar-typed PromQL expression to a scalar expression. A constant +/// sub-expression folds to one `Literal` (as `num_expr` always did); anything +/// else keeps its structure: `-time()` → `Negative`, `time() - 1` → +/// `Arithmetic`, `scalar(v)` → `PromqlScalarFromVector`, and a `bool` +/// comparison → `Case(Compare → 1, else 0)` (PromQL yields `0`/`1`). +fn lower_scalar(expr: &Expr) -> Result { + if let Ok(v) = num_expr(expr) { + return Ok(Scalar::Literal(ScalarValue::Float64(v))); + } + match expr { + Expr::Paren(p) => lower_scalar(&p.expr), + Expr::Unary(u) => Ok(Scalar::Negative { + expr: Box::new(lower_scalar(&u.expr)?), + semantics: PROMQL, + }), + Expr::Binary(bin) => lower_scalar_binary(bin), + Expr::Call(call) => match call.func.name { + "time" => Ok(Scalar::EvalTimestamp), + "pi" => Ok(Scalar::Literal(ScalarValue::Float64(std::f64::consts::PI))), + "scalar" => Ok(Scalar::PromqlScalarFromVector(Rc::new(walk(arg( + call, 0, + )?)?))), + // `min_of`/`max_of` fold only over constants (#89); the fold above + // failed, so surface its error for the non-constant argument. + name if is_scalar_reducer_fn(name) => Err(num_expr(expr).unwrap_err()), + other => Err(LoweringError::UnsupportedFunction(other.to_string())), + }, + other => Err(LoweringError::UnsupportedFeature(format!( + "scalar expression `{other}`" + ))), + } +} + +/// ` op `: arithmetic is an `Arithmetic` expression; a +/// comparison needs the `bool` modifier (PromQL has no scalar filter) and +/// becomes `Case(Compare → 1.0, else 0.0)`. The parser already rejects both a +/// bool-less scalar comparison and a scalar set op; both are re-checked here. +fn lower_scalar_binary(bin: &BinaryExpr) -> Result { + let left = Box::new(lower_scalar(&bin.lhs)?); + let right = Box::new(lower_scalar(&bin.rhs)?); + match binop(bin.op.id())? { + BinaryOpKind::Arithmetic(op) => Ok(Scalar::Arithmetic { + op, + left, + right, + semantics: PROMQL, + }), + BinaryOpKind::Compare(op) | BinaryOpKind::CompareBool(op) => { + if !bin.return_bool() { + return Err(LoweringError::InvalidParameter( + "a comparison between two scalars requires the `bool` modifier".into(), + )); + } + let compare = Scalar::Compare { + left, + op, + right, + semantics: PROMQL, + }; + Ok(Scalar::Case { + operand: None, + branches: vec![(compare, Scalar::Literal(ScalarValue::Float64(1.0)))], + else_expr: Some(Box::new(Scalar::Literal(ScalarValue::Float64(0.0)))), + }) + } + BinaryOpKind::Set(_) => Err(LoweringError::UnsupportedFeature( + "set operator between two scalars".into(), + )), + } +} + +/// A binary operation over two vectors. +fn vector_binary( + kind: BinaryOpKind, + vector_match: Option, + return_bool: bool, + lhs: Unresolved, + rhs: Unresolved, +) -> Unresolved { + Unresolved::BinaryOp { + operator: BinaryOperator { + kind, + vector_match, + checked_relative_division: false, + checked_finite_division: false, + }, + return_bool, + lhs: Rc::new(lhs), + rhs: Rc::new(rhs), + } +} + +/// Lower a bare function call (`rate(m[5m])`, `max_over_time(m[5m])`, …). +/// +/// The common case routes through the flat `lower_inner_call` template. The one +/// exception is a `*_over_time`/`quantile_over_time` function applied to a +/// **sub-query** (`max_over_time(rate(m[5m])[1h:])`): its argument is a +/// `PromQLSubquery`, not a matrix selector, so the flat template's +/// `extract_matrix` can't accept it. Lower the sub-query recursively and reduce +/// it per series (issue #27). +fn walk_call(call: &Call) -> Result { + if let Some(tree) = range_fn_over_subquery(call)? { + return Ok(tree); + } + build(lower_inner_call(call)?, vec![], Outer::None) +} + +/// A range-vector function applied to a **sub-query** — `f([range:res])`. +/// +/// Covers the whole range-vector family: the `*_over_time` reducers, +/// `rate`/`irate`/`increase`, and the counter-derivatives +/// (`changes`/`delta`/`idelta`/`deriv`/`resets`/`predict_linear`/ +/// `double_exponential_smoothing`). Each lowers to a per-series `Aggregate{[f]}` +/// directly over the `PromqlSubquery` — the sub-query is the range context, so +/// there is no separate `Window`/`TimeRange` (this walk treats the `PromqlSubquery` +/// node itself as the range marker). Returns `None` when `call` isn't a range +/// function or its argument isn't a sub-query, so the flat matrix-selector +/// template still handles `f(m[w])` (issues #42, #55). +fn range_fn_over_subquery(call: &Call) -> Result> { + // `rate`/`increase`/`irate` carry their window in the `AggFunc`; over a + // sub-query that window is the sub-query's own range. + if let "rate" | "irate" | "increase" = call.func.name { + let arg_expr = arg(call, 0)?; + if subquery_range(arg_expr).is_none() { + return Ok(None); + } + let inner = match call.func.name { + "rate" => InnerFunc::Rate, + "irate" => InnerFunc::IRate, + "increase" => InnerFunc::Increase, + _ => unreachable!(), + }; + return Ok(Some(per_series_aggregate( + vec![], + inner_intent(&inner), + walk(arg_expr)?, + ))); + } + + // `*_over_time` reducers + counter-derivatives: the func-kind, and the index + // of the matrix/sub-query argument (`quantile_over_time` reads φ from arg 0, + // so its matrix is arg 1; the rest take arg 0 + trailing scalar params). + let (inner, matrix_idx): (InnerFunc, usize) = match call.func.name { + "avg_over_time" => (InnerFunc::Avg, 0), + "min_over_time" => (InnerFunc::Min, 0), + "max_over_time" => (InnerFunc::Max, 0), + "sum_over_time" => (InnerFunc::Sum, 0), + "stddev_over_time" => (InnerFunc::StdDev, 0), + "stdvar_over_time" => (InnerFunc::Variance, 0), + "count_over_time" => (InnerFunc::Count, 0), + "distinct_over_time" => (InnerFunc::Cardinality, 0), + "entropy_over_time" => (InnerFunc::FrequencyEntropy, 0), + "l2_over_time" => (InnerFunc::FrequencyL2, 0), + "quantile_over_time" => (InnerFunc::Quantile(quantile_param(num_arg(call, 0)?)?), 1), + "changes" => (InnerFunc::Changes, 0), + "delta" => (InnerFunc::Delta, 0), + "idelta" => (InnerFunc::IDelta, 0), + "deriv" => (InnerFunc::Deriv, 0), + "resets" => (InnerFunc::Resets, 0), + "last_over_time" => (InnerFunc::LastOverTime, 0), + "first_over_time" => (InnerFunc::FirstOverTime, 0), + "mad_over_time" => (InnerFunc::MadOverTime, 0), + "ts_of_min_over_time" => (InnerFunc::TsOfMinOverTime, 0), + "ts_of_max_over_time" => (InnerFunc::TsOfMaxOverTime, 0), + "ts_of_first_over_time" => (InnerFunc::TsOfFirstOverTime, 0), + "ts_of_last_over_time" => (InnerFunc::TsOfLastOverTime, 0), + "predict_linear" => (InnerFunc::PredictLinear(num_arg(call, 1)?), 0), + "double_exponential_smoothing" => ( + InnerFunc::DoubleExp { + smoothing: num_arg(call, 1)?, + trend: num_arg(call, 2)?, + }, + 0, + ), + _ => return Ok(None), + }; + let arg_expr = arg(call, matrix_idx)?; + if !is_subquery(arg_expr) { + return Ok(None); + } + Ok(Some(per_series_aggregate( + vec![], + inner_intent(&inner), + walk(arg_expr)?, + ))) +} + +/// A (parenthesised) PromQL sub-query — `[range:res]`. +fn is_subquery(expr: &Expr) -> bool { + subquery_range(expr).is_some() +} + +/// The `range` of a (parenthesised) sub-query argument, if it is one. +fn subquery_range(expr: &Expr) -> Option { + match expr { + Expr::Subquery(sq) => Some(sq.range), + Expr::Paren(p) => subquery_range(&p.expr), + _ => None, + } +} + +fn walk_aggregate(agg: &AggregateExpr) -> Result { + let (keys, without) = resolve_group(agg)?; + let outer = outer_kind(agg)?; + + // `without(...)` grouping is modelled only for the reducing aggregations + // (sum/avg/count/…), whose grouping lives on an `Aggregate` node. `topk`/ + // `bottomk` (→ `Sort.partition_by`) and `limitk`/`limit_ratio` (→ `PromqlSeriesSample`) + // would need without-partitioning too; reject rather than silently lower + // them as a `by` grouping (issue #39). + if without && matches!(outer, Outer::TopK { .. } | Outer::Sample { .. }) { + return Err(LoweringError::UnsupportedFeature( + "`without(...)` is only supported on reducing aggregations, not \ + topk/bottomk/limitk" + .into(), + )); + } + + // Fast path — the argument is a bare selector or a single range-vector + // function (`rate`/`increase`/`*_over_time`). `lower_inner` lowers it via the + // flat selector/call template, which also recognises the heavy-hitter + // `topk(k, count_over_time(...))` shape. This is the common two-level case + // (`sum by (job) (rate(m[5m]))`). + // + // General nesting — the argument is itself a composite expression: another + // aggregate (`max(sum by (job) (rate(m[5m])))`), a binary op, a sub-query, or + // a function lowered elsewhere. Lower it recursively with the same `walk` + // used at the top level, then wrap it in the outer aggregation (issue #27; a + // negated argument `sum(-m)` lowers here too, #36). A genuinely unsupported + // inner expression surfaces its own error rather than being mislowered. + // + // Either way, `mark_without` flips the resulting outer `Aggregate` to the + // exclusion form when the modifier was `without(...)`. + let built = match lower_inner(&agg.expr) { + Ok(inner) => build(inner, keys, outer)?, + Err(_) => build_over_sub_dag(outer, keys, walk(&agg.expr)?)?, + }; + Ok(mark_without(built, without)) +} + +/// Map an `AggregateExpr`'s operator (`sum`/`avg`/`topk`/…) to the [`Outer`] +/// shape, independent of what the argument is — so both the flat fast path and +/// the general recursive path share one operator-dispatch. +fn outer_kind(agg: &AggregateExpr) -> Result { + let op = agg.op.id(); + + Ok(if op == token::T_TOPK { + Outer::TopK { + k: count_param(agg)?, + descending: true, + } + } else if op == token::T_BOTTOMK { + Outer::TopK { + k: count_param(agg)?, + descending: false, + } + } else if op == token::T_COUNT { + Outer::Count + } else if op == token::T_SUM { + Outer::Plain(OuterIntent::Sum) + } else if op == token::T_GROUP { + // `group(v)` yields a constant 1 per group (presence), not a sum of + // values — a distinct intent, never folded onto `Sum` (issue #49). + Outer::Plain(OuterIntent::Group) + } else if op == token::T_COUNT_VALUES { + // `count_values("l", v)` groups by sample value and counts, emitting the + // value as a new label `l` (the string parameter) — issue #49. + Outer::CountValues { + label: str_param(agg)?, + } + } else if op == token::T_LIMITK { + // `limitk(k, v)` — up to k series per group (issue #86). + Outer::Sample { + kind: SampleKind::LimitK(count_param(agg)? as usize), + } + } else if op == token::T_LIMIT_RATIO { + // `limit_ratio(r, v)` — an r-fraction of series per group (issue #86). + Outer::Sample { + kind: SampleKind::LimitRatio(ratio_param(agg)?), + } + } else if op == token::T_AVG { + Outer::Plain(OuterIntent::Avg) + } else if op == token::T_MIN { + Outer::Plain(OuterIntent::Min) + } else if op == token::T_MAX { + Outer::Plain(OuterIntent::Max) + } else if op == token::T_STDDEV { + Outer::Plain(OuterIntent::StdDev) + } else if op == token::T_STDVAR { + Outer::Plain(OuterIntent::Variance) + } else if op == token::T_QUANTILE { + Outer::Plain(OuterIntent::Quantile(quantile_param(num_param(agg)?)?)) + } else { + return Err(LoweringError::UnsupportedAggregateOp(format!( + "aggregate token {op}" + ))); + }) +} + +/// Wrap an already-lowered Unresolved sub-DAG in the outer aggregation. This is the +/// general-nesting counterpart to [`build`]: where `build` assembles the +/// two-level shape from a flat [`Inner`], this composes the outer operator over +/// an arbitrary child (`max(sum by (job) (…))`, `sum(a + b)`, …). +/// +/// A heavy-hitter `TopK` is only recognised on the flat `count_over_time` shape +/// (handled in `build`); over a general sub-DAG, `topk`/`bottomk` is a generic +/// order-by-value + limit — the same `Sort{partition_by} → Limit` pair `build` +/// emits for any non-heavy-hitter ranking. +/// Flip the outer `Aggregate` produced for a `without(...)` grouping into the +/// exclusion form. The reducing-aggregation `build` paths place that aggregate +/// at the root; `walk_aggregate` has already rejected the non-aggregate outers +/// (topk/limitk), so a `without` grouping always has an `Aggregate` here (issue +/// #39). A no-op when the modifier was `by`. +/// Flip the outer `Aggregate` produced for a `without(...)` grouping into the +/// exclusion form. A no-op when the modifier was `by`. +/// +/// `reduction_for` (used by [`windowed_aggregate`]/[`outer_aggregate`] to +/// build this node) decides `PerEntity` vs `Reduce(by)` *without* knowing +/// about `without` yet — it only ever sees `by`-mode keys, since `without`'s +/// excluded-labels list is applied here, after the fact, exactly like the +/// pre-#179 legacy relational tree's own `mark_without` did (its +/// converter read `without` only after this front-end step had already set +/// it). Whether +/// `reduction_for` picked `PerEntity` (only possible when `keys` was empty) +/// or `Reduce(by)`, the correct answer under `without(...)` is always +/// `Reduce(without(keys))`: a `without` grouping is never label-preserving — +/// per-entity requires `!by.is_without()` — so this both re-tags an existing +/// `Reduce` and upgrades a wrongly-early `PerEntity` guess, uniformly. +fn mark_without(tree: Unresolved, without: bool) -> Unresolved { + if !without { + return tree; + } + match tree { + Unresolved::Aggregate { + reduction, + measures, + output_names, + filters, + having, + child, + } => { + let keys = match reduction { + Reduction::Reduce(by) => by.keys().to_vec(), + Reduction::PerEntity => vec![], + }; + Unresolved::Aggregate { + reduction: Reduction::Reduce(GroupKeys::without(keys)), + measures, + output_names, + filters, + having, + child, + } + } + other => other, + } +} + +fn build_over_sub_dag(outer: Outer, keys: Vec, child: Unresolved) -> Result { + Ok(match outer { + // `walk_aggregate` always passes a real aggregator; `None` can't occur. + Outer::None => child, + Outer::Plain(intent) => outer_aggregate(keys, outer_intent(&intent), child), + Outer::Count => outer_aggregate(keys, count(), child), + Outer::CountValues { label } => { + outer_aggregate(keys, AggIntent::CountValues { label }, child) + } + Outer::Sample { kind } => Unresolved::PromqlSeriesSample { + by: keys.into(), + kind, + child: Rc::new(child), + }, + Outer::TopK { k, descending } => { + let weighted_counter_ranking = matches!( + &child, + Unresolved::Aggregate { + measures, + child: sum_child, + .. + } if matches!(measures.as_slice(), [AggIntent::Sum { .. }]) + && matches!(sum_child.as_ref(), Unresolved::Aggregate { measures, .. } + if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])) + ); + let direct_counter_ranking = matches!(&child, Unresolved::Aggregate { + measures, reduction: Reduction::PerEntity, .. + } if matches!(measures.as_slice(), [AggIntent::Rate | AggIntent::Increase])); + if descending && (weighted_counter_ranking || direct_counter_ranking) { + return Ok(outer_aggregate( + keys, + AggIntent::TopK { + k: k as usize, + accuracy: current_accuracy(), + }, + child, + )); + } + ranked_by_value(keys, k, descending, child) + } + }) +} + +/// Generic `topk`/`bottomk`: `Limit{k} → Sort{value, partition_by: keys}` over +/// `child` — an order-by-value ranking, not a heavy-hitter intent. +fn ranked_by_value( + keys: Vec, + k: u64, + descending: bool, + child: Unresolved, +) -> Unresolved { + let sorted = Unresolved::Sort { + keys: vec![UnresolvedSortKey { + expr: Scalar::Column(ColumnRef::SampleValue), + ascending: !descending, + nulls_first: false, + }], + partition_by: keys.into(), + child: Rc::new(child), + }; + Unresolved::Limit { + n: Some(k as usize), + offset: 0, + partition_by: GroupKeys::none(), + child: Rc::new(sorted), + } +} + +/// The `histogram_*` function family (issues #43, histogram_quantile). +/// +/// `histogram_quantile(φ, )` lowers `` in full — preserving any +/// `sum by (le)` / `rate` structure inside it. The classic `le`-bucket form +/// becomes [`classic_histogram_quantile`]; a native histogram or raw samples +/// become a `Quantile` over the whole argument. +/// The native-histogram accessors (`histogram_count`/`sum`/`avg`/`stddev`/ +/// `stdvar`/`fraction`) each extract one float per series, lowering to a +/// per-series `Aggregate{[accessor]}` directly over the (instant) argument. +/// `histogram_fraction(lower, upper, v)` reads its bounds from args 0/1 and the +/// vector from arg 2; the rest take the vector at arg 0. +fn walk_histogram(call: &Call) -> Result { + if call.func.name == "histogram_quantiles" { + return walk_histogram_quantiles(call); + } + if call.func.name == "histogram_quantile" { + let phi = quantile_param(num_arg(call, 0)?)?; + let arg_expr = arg(call, 1)?; + // Two lowerings of `histogram_quantile(φ, …)`: + // - classic `le`-bucket form → `HistogramQuantile`, exact interpolation + // over cumulative buckets (not sketch-able). + // - native-histogram / raw-samples form → the generic `Quantile` intent + // (sketch-able). + // The true signal is the argument's sample type: a declared + // `HistogramKind` (issue #79) drives the choice when available, else we + // fall back to the structural `by (le)`/`_bucket` heuristic (issue #43). + if !histogram_arg_is_sketchable(arg_expr)? { + return Ok(classic_histogram_quantile(phi, "", walk(arg_expr)?)); + } + let func = AggIntent::Quantile { + col: None, + q: phi, + accuracy: current_accuracy(), + }; + return Ok(outer_aggregate(vec![], func, walk(arg_expr)?)); + } + Err(LoweringError::UnsupportedFeature( + "native histogram samples have no IR representation".into(), + )) +} + +/// Classic-bucket `histogram_quantile(φ, child)`. One histogram is the set of +/// series that differ only in `le`, so the aggregate groups `without (le)`. +/// That grouping also seeds `le` into a usage-derived source schema, even +/// when no matcher names it. An empty `output_name` keeps the intent-keyed name. +fn classic_histogram_quantile(q: f64, output_name: &str, child: Unresolved) -> Unresolved { + let le = ColumnRef::Named("le".into()); + Unresolved::Aggregate { + reduction: Reduction::Reduce(GroupKeys::without(vec![le.clone()])), + measures: vec![AggIntent::HistogramQuantile { q, le }], + output_names: vec![output_name.into()], + filters: vec![], + having: None, + child: Rc::new(child), + } +} + +/// `histogram_quantiles(v, "label", φ₀, φ₁, …)` — the experimental multi-quantile +/// form (issue #109). It is `histogram_quantile(φᵢ, v)` fanned out over the +/// quantiles, each branch's output series tagged with `label = φᵢ`. +/// +/// Lowers to a `Concat` of one `PromqlRelabel`-wrapped quantile branch per φ, reusing +/// the single-quantile decision — classic `le`-buckets interpolate +/// (`HistogramQuantile`), native histograms / raw samples take the sketch-able +/// `Quantile` (issues #43 / #79) — so the two functions cannot diverge. +/// +/// The vector argument is lowered once per branch, duplicating the sub-DAG — +/// a future workload-level reuse pass could hoist it back into a single +/// producer. +/// +/// Each branch aliases its value column to `value` rather than taking the +/// intent-keyed name (`quantile_0_5`, `quantile_0_9`, …). `Concat` derives its +/// schema from the first child, so branches that disagree on a column *name* +/// would make the merged schema silently misdescribe every branch but one. The +/// quantile is carried by the `label` column, which is exactly where Prometheus +/// puts it. +fn walk_histogram_quantiles(call: &Call) -> Result { + let vec_expr = arg(call, 0)?; + let label = str_arg(call, 1)?; + if call.args.args.len() < 3 { + return Err(LoweringError::MissingArgument( + "histogram_quantiles(v, label, φ…) needs at least one quantile".into(), + )); + } + // The bucket-vs-native choice is a property of the argument, not of φ. + let sketchable = histogram_arg_is_sketchable(vec_expr)?; + let branches = (2..call.args.args.len()) + .map(|i| { + let phi = bounded_quantile_param(num_arg(call, i)?)?; + let child = walk(vec_expr)?; + // Each branch aliases its value column to "value" (not the + // intent-keyed default) so `Concat` — which derives its schema + // from the first branch — doesn't silently misdescribe the rest. + let quantile = if sketchable { + let intent = AggIntent::Quantile { + col: None, + q: phi, + accuracy: current_accuracy(), + }; + Unresolved::Aggregate { + reduction: reduction_for(&[], intent.is_per_series()), + measures: vec![intent], + output_names: vec!["value".into()], + filters: vec![], + having: None, + child: Rc::new(child), + } + } else { + classic_histogram_quantile(phi, "value", child) + }; + Ok(Unresolved::PromqlRelabel { + dst: label.clone(), + value: Scalar::Literal(ScalarValue::Utf8(open_metrics_float(phi))), + child: Rc::new(quantile), + }) + }) + .collect::>>()?; + // No discriminator asserted here today (issue #228): the φ value each + // branch carries via `PromqlRelabel` *is* structurally a distinct + // per-branch discriminator, but nothing downstream currently needs the + // resulting compound unique key — see + // `docs/design_docs/concat-unique-keys-decision.md`. `Unresolved::concat` + // keeps `output_schema`'s default (drop `unique_keys` entirely). + Ok(Unresolved::concat(branches)) +} + +/// Prometheus's `labels.FormatOpenMetricsFloat` — how `histogram_quantiles` +/// renders each φ into its label value. Go's `%g` shortest round-trip, switching +/// to exponent form outside `[1e-4, 1e21)`, with `.0` appended when the result +/// would otherwise look like an integer. +fn open_metrics_float(v: f64) -> String { + // The cases upstream hardcodes. + if v == 1.0 { + return "1.0".into(); + } + if v == 0.0 { + return "0.0".into(); + } + if v == -1.0 { + return "-1.0".into(); + } + if v.is_nan() { + return "NaN".into(); + } + if v.is_infinite() { + return if v.is_sign_positive() { "+Inf" } else { "-Inf" }.into(); + } + let sci = format!("{v:e}"); + let exp: i32 = sci + .split_once('e') + .and_then(|(_, e)| e.parse().ok()) + .unwrap_or(0); + if !(-4..21).contains(&exp) { + // Go writes a signed, zero-padded two-digit exponent: `1e-05`. + let (mantissa, _) = sci.split_once('e').unwrap_or((sci.as_str(), "0")); + let sign = if exp < 0 { '-' } else { '+' }; + return format!("{mantissa}e{sign}{:02}", exp.abs()); + } + let s = format!("{v}"); + if s.contains(['e', '.']) { + s + } else { + format!("{s}.0") + } +} + +/// The calendar functions (issue #46); `time()` is scalar-typed and lowers in +/// `lower_scalar`. +fn is_time_fn(name: &str) -> bool { + matches!( + name, + "timestamp" + | "minute" + | "hour" + | "day_of_week" + | "day_of_month" + | "day_of_year" + | "month" + | "year" + | "days_in_month" + ) +} + +/// `timestamp(v)` and the calendar accessors → `Aggregate{[TimeFn(f)]}` over +/// the argument vector, or over `PromqlVectorFromScalar(EvalTimestamp)` for the +/// no-argument calendar forms (`hour()`, `day_of_week()`, …). Issue #46. +fn walk_time(call: &Call) -> Result { + // timestamp() reads the selected sample's timestamp, not its value. + if call.func.name == "timestamp" { + return Ok(outer_aggregate( + vec![], + AggIntent::TimeFn(TimeFunc::Timestamp), + walk(arg(call, 0)?)?, + )); + } + let child = if call.args.args.is_empty() { + Unresolved::PromqlVectorFromScalar(Scalar::EvalTimestamp) + } else { + walk(arg(call, 0)?)? + }; + Ok(Unresolved::PromqlMap { + child: Rc::new(child), + sample: Scalar::FunctionCall { + name: format!("promql_{}", call.func.name), + args: vec![Scalar::Column(ColumnRef::SampleValue)], + }, + drop_metric_name: true, + }) +} + +/// The presence functions (issue #47). +fn is_presence_fn(name: &str) -> bool { + matches!(name, "absent" | "absent_over_time" | "present_over_time") +} + +/// `absent(v)` / `absent_over_time(m[w])` / `present_over_time(m[w])` — lowered +/// to an `Aggregate{[Absent/…]}` over the (instant or range) argument. The +/// empty-result → synthesized-1-sample logic is a post-ASAP/runtime concern; +/// the canonical tree only marks the operation (issue #47). +fn walk_presence(call: &Call) -> Result { + let func = match call.func.name { + "absent" => AggIntent::Absent, + "absent_over_time" => AggIntent::AbsentOverTime, + "present_over_time" => AggIntent::PresentOverTime, + other => return Err(LoweringError::UnsupportedFunction(other.to_string())), + }; + // arg 0 is the instant vector (`absent`) or range vector (`*_over_time`); + // `walk` produces a `Window` for the matrix-selector forms. + Ok(outer_aggregate(vec![], func, walk(arg(call, 0)?)?)) +} + +/// The scalar→vector conversion (issue #48); `scalar(v)` is scalar-typed and +/// lowers in `lower_scalar`. `info` is *not* here: it is a label-enrichment +/// join, not a type conversion (#84). +fn is_typeconv_fn(name: &str) -> bool { + name == "vector" +} + +/// `vector(s)` — promote a scalar to a label-less instant vector carrying the +/// scalar expression `s` (issue #48). +fn walk_typeconv(call: &Call) -> Result { + Ok(Unresolved::PromqlVectorFromScalar(lower_scalar(arg( + call, 0, + )?)?)) +} + +/// The instant-vector reordering functions (issue #51). +fn is_sort_fn(name: &str) -> bool { + matches!( + name, + "sort" | "sort_desc" | "sort_by_label" | "sort_by_label_desc" + ) +} + +/// `sort`/`sort_desc(v)` reorder an instant vector by sample value; +/// `sort_by_label`/`sort_by_label_desc(v, "l"…)` reorder by label values. All +/// lower to a bare `Sort` (no `Limit`) over the vector argument — a faithful, +/// row-preserving reordering (issue #51). +fn walk_sort(call: &Call) -> Result { + let child = Rc::new(walk(arg(call, 0)?)?); + let (by_value, ascending) = match call.func.name { + "sort" => (true, true), + "sort_desc" => (true, false), + "sort_by_label" => (false, true), + "sort_by_label_desc" => (false, false), + other => return Err(LoweringError::UnsupportedFunction(other.to_string())), + }; + let sort_key = |expr| UnresolvedSortKey { + expr, + ascending, + nulls_first: false, + }; + let keys = if by_value { + vec![sort_key(Scalar::Column(ColumnRef::SampleValue))] + } else { + // `sort_by_label(v, "l1", "l2", …)` — one key per label arg, in order. + if call.args.args.len() < 2 { + return Err(LoweringError::MissingArgument( + "sort_by_label needs at least one label".into(), + )); + } + (1..call.args.args.len()) + .map(|i| { + Ok(sort_key(Scalar::Column(ColumnRef::Named(str_arg( + call, i, + )?)))) + }) + .collect::>>()? + }; + Ok(Unresolved::Sort { + keys, + partition_by: GroupKeys::none(), + child, + }) +} + +/// `info(v, [selector])` — a label-enrichment join. Lowers the input vector and +/// wraps it in an `PromqlInfoEnrich` carrying the (optional) data-label selector's +/// matchers; the actual join against the info metric — on shared identifying +/// labels — is resolved during post-ASAP binding (issue #84). +fn walk_info(call: &Call) -> Result { + let child = Rc::new(walk(arg(call, 0)?)?); + let selector = match call.args.args.get(1) { + Some(sel) => info_selector(sel)?, + None => Vec::new(), // default: enrich from `target_info` + }; + Ok(Unresolved::PromqlInfoEnrich { selector, child }) +} + +/// Extract the `info` data-label selector's matchers. Unlike an ordinary +/// selector these are **info-metric-side** and may carry regex / multiple +/// `__name__` matchers (which pick the info metric(s)), so they bypass the +/// single-metric `vs_parts` restriction and are kept symbolic. +fn info_selector(expr: &Expr) -> Result> { + match expr { + Expr::VectorSelector(vs) => Ok(vs + .matchers + .matchers + .iter() + .map(|m| InfoMatcher { + label: m.name.clone(), + op: match &m.op { + MatchOp::Equal => CompareOpKind::Eq, + MatchOp::NotEqual => CompareOpKind::Ne, + MatchOp::Re(_) => CompareOpKind::Regex, + MatchOp::NotRe(_) => CompareOpKind::NotRegex, + }, + value: m.value.clone(), + }) + .collect()), + Expr::Paren(p) => info_selector(&p.expr), + other => Err(LoweringError::UnsupportedFeature(format!( + "`info` data-label selector must be a label-matcher set, got `{other}`" + ))), + } +} + +/// The label-rewrite functions (issue #50). +fn is_label_fn(name: &str) -> bool { + matches!(name, "label_replace" | "label_join") +} + +/// `label_replace(v, dst, replacement, src, regex)` / +/// `label_join(v, dst, sep, src…)` — per-series label rewrites. Both lower to a +/// `PromqlRelabel` over the fully-lowered vector argument, differing only in the +/// expression that computes the destination label: `label_replace` a regex +/// capture-expansion, `label_join` a separator-joined concatenation. Sample +/// values are untouched; the regex-match-or-passthrough and capture-expansion +/// are post-ASAP/runtime concerns (issue #50). +fn walk_label(call: &Call) -> Result { + let child = Rc::new(walk(arg(call, 0)?)?); + match call.func.name { + "label_replace" => { + let dst = str_arg(call, 1)?; + let replacement = str_arg(call, 2)?; + let src = str_arg(call, 3)?; + let regex = str_arg(call, 4)?; + let value = Scalar::FunctionCall { + name: "label_replace".into(), + args: vec![ + Scalar::Column(ColumnRef::Named(src)), + Scalar::Literal(ScalarValue::Utf8(regex)), + Scalar::Literal(ScalarValue::Utf8(replacement)), + ], + }; + Ok(Unresolved::PromqlRelabel { dst, value, child }) + } + "label_join" => { + // label_join(v, dst, sep, src_1, …, src_n) — needs ≥1 source label. + if call.args.args.len() < 4 { + return Err(LoweringError::MissingArgument( + "label_join(v, dst, sep, src…) needs at least one source label".into(), + )); + } + let dst = str_arg(call, 1)?; + let sep = str_arg(call, 2)?; + let mut args = vec![Scalar::Literal(ScalarValue::Utf8(sep))]; + for i in 3..call.args.args.len() { + args.push(Scalar::Column(ColumnRef::Named(str_arg(call, i)?))); + } + let value = Scalar::FunctionCall { + name: "label_join".into(), + args, + }; + Ok(Unresolved::PromqlRelabel { dst, value, child }) + } + other => Err(LoweringError::UnsupportedFunction(other.to_string())), + } +} + +/// The element-wise math / trig functions (issue #45). +fn is_math_fn(name: &str) -> bool { + matches!( + name, + "abs" + | "ceil" + | "floor" + | "exp" + | "ln" + | "log2" + | "log10" + | "sqrt" + | "sgn" + | "sin" + | "cos" + | "tan" + | "asin" + | "acos" + | "atan" + | "sinh" + | "cosh" + | "tanh" + | "asinh" + | "acosh" + | "atanh" + | "deg" + | "rad" + | "round" + | "clamp" + | "clamp_min" + | "clamp_max" + ) +} + +/// A math / trig function — a per-series element-wise value transform, lowered +/// to a typed scalar projection over the instant-vector argument. +/// `pi()` is scalar-typed and lowers in `lower_scalar` (issue #45). +fn walk_math(call: &Call) -> Result { + let mut args = vec![Scalar::Column(ColumnRef::SampleValue)]; + for index in 1..call.args.args.len() { + args.push(lower_scalar(arg(call, index)?)?); + } + if call.func.name == "round" && args.len() == 1 { + args.push(Scalar::Literal(ScalarValue::Float64(1.0))); + } + Ok(Unresolved::PromqlMap { + child: Rc::new(walk(arg(call, 0)?)?), + sample: Scalar::FunctionCall { + name: format!("promql_{}", call.func.name), + args, + }, + drop_metric_name: true, + }) +} + +/// Whether `expr` is a **classic cumulative-bucket** `histogram_quantile` +/// argument — as opposed to a native histogram or raw samples. Recognised +/// structurally, by any of: +/// - a `by (le)` grouping (`sum by (le) (…)`), +/// - a selector on a classic `_bucket` metric (`http_request_…_bucket`), +/// - a selector with an `le` label matcher (`{le="…"}`). +/// +/// The bucket form must be *interpolated* (`HistogramQuantile`); everything +/// else is a sketch-able generic `Quantile`. This is a heuristic proxy for the +/// real signal — the argument's sample type — which isn't visible at lowering; +/// see the follow-up issue on the discrimination criteria (issue #43). +/// Whether `histogram_quantile(φ, arg)` lowers to the sketch-able generic +/// `Quantile` (`true`) or exact classic-bucket interpolation (`false`). +/// +/// Metadata wins: if any metric referenced in `arg` has a declared +/// [`HistogramKind`](crate::unified::histogram::HistogramKind), that decides it (issue +/// #79) — this fixes both the false-positive (a `…_bucket`-named non-histogram +/// declared `RawSamples`) and the false-negative (a suffix-less classic +/// histogram declared `ClassicBucket`) of the structural heuristic. With no +/// declaration, fall back to the structural `by (le)`/`_bucket` heuristic. +fn histogram_arg_is_sketchable(arg: &Expr) -> Result { + let mut metrics = Vec::new(); + collect_metric_names(arg, &mut metrics); + let kinds = metrics + .iter() + .filter_map(|metric| crate::unified::histogram::current_kind_of(metric)) + .collect::>(); + if kinds.contains(&crate::unified::histogram::HistogramKind::Native) { + return Err(LoweringError::UnsupportedFeature( + "native histogram samples have no IR representation".into(), + )); + } + if let Some(kind) = kinds.first() { + if kinds.iter().any(|other| other != kind) { + return Err(LoweringError::UnsupportedFeature( + "mixed histogram sample contracts".into(), + )); + } + return Ok(kind.is_sketchable()); + } + if is_classic_bucket_arg(arg) { + Ok(false) + } else { + Err(LoweringError::UnsupportedFeature("histogram_quantile requires classic buckets; use quantile for float samples or explicitly declare the RawSamples extension".into())) + } +} + +/// Collect the metric names of every vector/matrix selector reachable in `expr` +/// (for the metadata lookup in [`histogram_arg_is_sketchable`]). Skips +/// name-less selectors like `{le="…"}`. +fn collect_metric_names(expr: &Expr, out: &mut Vec) { + match expr { + Expr::VectorSelector(vs) => { + if let Ok((metric, ..)) = vs_parts(vs) { + if !metric.is_empty() { + out.push(metric); + } + } + } + Expr::MatrixSelector(ms) => { + if let Ok((metric, ..)) = vs_parts(&ms.vs) { + if !metric.is_empty() { + out.push(metric); + } + } + } + Expr::Paren(p) => collect_metric_names(&p.expr, out), + Expr::Unary(u) => collect_metric_names(&u.expr, out), + Expr::Subquery(s) => collect_metric_names(&s.expr, out), + Expr::Aggregate(a) => collect_metric_names(&a.expr, out), + Expr::Binary(b) => { + collect_metric_names(&b.lhs, out); + collect_metric_names(&b.rhs, out); + } + Expr::Call(c) => c + .args + .args + .iter() + .for_each(|a| collect_metric_names(a, out)), + _ => {} + } +} + +fn is_classic_bucket_arg(expr: &Expr) -> bool { + match expr { + Expr::Paren(p) => is_classic_bucket_arg(&p.expr), + Expr::Unary(u) => is_classic_bucket_arg(&u.expr), + Expr::Subquery(s) => is_classic_bucket_arg(&s.expr), + Expr::Aggregate(agg) => { + matches!( + &agg.modifier, + Some(LabelModifier::Include(ls)) if ls.labels.iter().any(|l| l == "le") + ) || is_classic_bucket_arg(&agg.expr) + } + Expr::Binary(b) => is_classic_bucket_arg(&b.lhs) || is_classic_bucket_arg(&b.rhs), + Expr::Call(c) => c.args.args.iter().any(|a| is_classic_bucket_arg(a)), + Expr::VectorSelector(vs) => selector_is_bucket(vs), + Expr::MatrixSelector(ms) => selector_is_bucket(&ms.vs), + _ => false, + } +} + +/// A classic histogram bucket selector — a `_bucket`-named metric (via bare name +/// or `__name__` matcher) or an explicit `le` label matcher. +fn selector_is_bucket(vs: &VectorSelector) -> bool { + let name = vs.name.as_deref().or_else(|| { + vs.matchers + .matchers + .iter() + .find(|m| m.name == "__name__") + .map(|m| m.value.as_str()) + }); + name.is_some_and(|n| n.ends_with("_bucket")) + || vs.matchers.matchers.iter().any(|m| m.name == "le") +} + +/// A binary op with at least one vector operand (a scalar/scalar op is +/// scalar-typed and never reaches here). A scalar side lowers to a +/// scalar expression; mixed operations resolve to Project or Filter. +fn walk_binary(bin: &BinaryExpr) -> Result { + let op = binop(bin.op.id())?; + let scalar_left = bin.lhs.value_type() == ValueType::Scalar; + if scalar_left || bin.rhs.value_type() == ValueType::Scalar { + let (scalar, vector) = if scalar_left { + (&bin.lhs, &bin.rhs) + } else { + (&bin.rhs, &bin.lhs) + }; + return Ok(Unresolved::PromqlScalarOp { + child: Rc::new(walk(vector)?), + scalar: lower_scalar(scalar)?, + op, + scalar_left, + return_bool: bin.return_bool(), + }); + } + let lhs = walk(&bin.lhs)?; + let rhs = walk(&bin.rhs)?; + // `VectorMatch` has no fill field; dropping fill would change which series + // are emitted and their values, so the query must fall back to exact + // execution instead. + if let Some(m) = &bin.modifier { + if m.fill_values.lhs.is_some() || m.fill_values.rhs.is_some() { + return Err(LoweringError::UnsupportedFeature(format!( + "`fill` vector-matching modifier: `{bin}`" + ))); + } + } + let vector_match = bin.modifier.as_ref().map(|m| { + let (kind, labels) = match &m.matching { + Some(LabelModifier::Include(ls)) => (VectorMatchKind::On, ls.labels.clone()), + Some(LabelModifier::Exclude(ls)) => (VectorMatchKind::Ignoring, ls.labels.clone()), + // No explicit `on(…)`/`ignoring(…)` — the parser attaches a default + // modifier to every set op (`and`/`or`/`unless`). The default is + // "match on all shared labels", which is exactly `ignoring([])` + // (ignore no labels). Representing it as `Ignoring([])` — not + // `On([])` — keeps it distinct from an explicit `on()` (match on the + // empty label set) while making it correctly equal to an explicit + // `ignoring()` (issue #68). + None => (VectorMatchKind::Ignoring, vec![]), + }; + let grouping = match &m.card { + VectorMatchCardinality::ManyToOne(ls) => Some(VectorGrouping { + side: GroupSide::Left, + labels: ls.labels.clone(), + }), + VectorMatchCardinality::OneToMany(ls) => Some(VectorGrouping { + side: GroupSide::Right, + labels: ls.labels.clone(), + }), + _ => None, + }; + VectorMatch { + kind, + labels, + grouping, + } + }); + Ok(vector_binary(op, vector_match, bin.return_bool(), lhs, rhs)) +} + +fn lower_inner(expr: &Expr) -> Result { + match expr { + Expr::VectorSelector(vs) => { + let (metric, matchers, shift) = vs_parts(vs)?; + Ok(Inner { + metric, + matchers, + window: None, + func: None, + shift, + }) + } + Expr::MatrixSelector(ms) => { + let (metric, matchers, shift) = vs_parts(&ms.vs)?; + Ok(Inner { + metric, + matchers, + window: Some(ms.range), + func: None, + shift, + }) + } + Expr::Paren(p) => lower_inner(&p.expr), + Expr::Call(call) => lower_inner_call(call), + other => Err(LoweringError::UnsupportedFeature(format!( + "aggregate argument: `{other}`" + ))), + } +} + +fn lower_inner_call(call: &Call) -> Result { + let name = call.func.name; + let at0 = |func: InnerFunc| -> Result { + let (metric, matchers, window, shift) = extract_matrix(arg(call, 0)?)?; + Ok(Inner { + metric, + matchers, + window: Some(window), + func: Some(func), + shift, + }) + }; + match name { + "rate" | "irate" => { + let (metric, matchers, window, shift) = extract_matrix(arg(call, 0)?)?; + Ok(Inner { + metric, + matchers, + window: Some(window), + func: Some(if name == "irate" { + InnerFunc::IRate + } else { + InnerFunc::Rate + }), + shift, + }) + } + "increase" => { + let (metric, matchers, window, shift) = extract_matrix(arg(call, 0)?)?; + Ok(Inner { + metric, + matchers, + window: Some(window), + func: Some(InnerFunc::Increase), + shift, + }) + } + "quantile_over_time" => { + let phi = quantile_param(num_arg(call, 0)?)?; + let (metric, matchers, window, shift) = extract_matrix(arg(call, 1)?)?; + Ok(Inner { + metric, + matchers, + window: Some(window), + func: Some(InnerFunc::Quantile(phi)), + shift, + }) + } + "avg_over_time" => at0(InnerFunc::Avg), + "min_over_time" => at0(InnerFunc::Min), + "max_over_time" => at0(InnerFunc::Max), + "sum_over_time" => at0(InnerFunc::Sum), + "stddev_over_time" => at0(InnerFunc::StdDev), + "stdvar_over_time" => at0(InnerFunc::Variance), + "count_over_time" => at0(InnerFunc::Count), + "distinct_over_time" => at0(InnerFunc::Cardinality), + "entropy_over_time" => at0(InnerFunc::FrequencyEntropy), + "l2_over_time" => at0(InnerFunc::FrequencyL2), + // Counter-derivative range functions (issue #44). Each has its own + // intent — `changes` (value-change count) and `resets` (counter-reset + // count) are NOT sample counts, so they are not aliased to + // `count_over_time`. The window is arg 0's matrix; scalar params follow. + "changes" => at0(InnerFunc::Changes), + "delta" => at0(InnerFunc::Delta), + "idelta" => at0(InnerFunc::IDelta), + "deriv" => at0(InnerFunc::Deriv), + "resets" => at0(InnerFunc::Resets), + // Additional range-vector reducers (issue #51) — same windowed + // per-series shape as the `*_over_time` family above. + "last_over_time" => at0(InnerFunc::LastOverTime), + "first_over_time" => at0(InnerFunc::FirstOverTime), + "mad_over_time" => at0(InnerFunc::MadOverTime), + "ts_of_min_over_time" => at0(InnerFunc::TsOfMinOverTime), + "ts_of_max_over_time" => at0(InnerFunc::TsOfMaxOverTime), + "ts_of_first_over_time" => at0(InnerFunc::TsOfFirstOverTime), + "ts_of_last_over_time" => at0(InnerFunc::TsOfLastOverTime), + "predict_linear" => { + let (metric, matchers, window, shift) = extract_matrix(arg(call, 0)?)?; + let seconds = num_arg(call, 1)?; + Ok(Inner { + metric, + matchers, + window: Some(window), + func: Some(InnerFunc::PredictLinear(seconds)), + shift, + }) + } + "double_exponential_smoothing" => { + let (metric, matchers, window, shift) = extract_matrix(arg(call, 0)?)?; + let smoothing = num_arg(call, 1)?; + let trend = num_arg(call, 2)?; + Ok(Inner { + metric, + matchers, + window: Some(window), + func: Some(InnerFunc::DoubleExp { smoothing, trend }), + shift, + }) + } + other => Err(LoweringError::UnsupportedFunction(other.to_string())), + } +} + +/// Assemble the Layer-2 tree from a lowered inner vector, the resolved group +/// keys, and the enclosing aggregator shape. +fn build(inner: Inner, keys: Vec, outer: Outer) -> Result { + match outer { + Outer::None => match &inner.func { + None => Ok(instant_source(inner.metric, inner.matchers, inner.shift)), + Some(f) => { + let intent = inner_intent(f); + Ok(windowed_aggregate(inner, keys, intent)) + } + }, + // An OUTER aggregation operator (`sum`/`avg`/…/`count`) over an inner + // range-vector function (`rate`/`increase`/`*_over_time`) is a + // two-level reduction: the inner func runs per series, the outer op + // then aggregates across series. Collapsing them into one aggregate + // silently drops a level — e.g. `sum(rate(m[w]))` must keep the `sum`. + Outer::Plain(intent) => Ok(match &inner.func { + None => windowed_aggregate(inner, keys, outer_intent(&intent)), + Some(f) => { + let inner_i = inner_intent(f); + let inner_agg = windowed_aggregate(inner, vec![], inner_i); + outer_aggregate(keys, outer_intent(&intent), inner_agg) + } + }), + Outer::Count => Ok(match &inner.func { + None => windowed_aggregate(inner, keys, count()), + Some(f) => { + let inner_i = inner_intent(f); + let inner_agg = windowed_aggregate(inner, vec![], inner_i); + outer_aggregate(keys, count(), inner_agg) + } + }), + Outer::CountValues { label } => Ok(match &inner.func { + None => windowed_aggregate(inner, keys, AggIntent::CountValues { label }), + Some(f) => { + let inner_i = inner_intent(f); + let inner_agg = windowed_aggregate(inner, vec![], inner_i); + outer_aggregate(keys, AggIntent::CountValues { label }, inner_agg) + } + }), + Outer::Sample { kind } => { + // Series sampling selects whole series unchanged — like generic + // `topk`, a range-vector argument reduces per series first (label- + // preserving), a bare selector is sampled directly; neither is + // wrapped in a reducing aggregate (issue #86). + let base = match inner.func.as_ref().map(inner_intent) { + Some(intent) => windowed_aggregate(inner, vec![], intent), + None => instant_source(inner.metric, inner.matchers, inner.shift), + }; + Ok(Unresolved::PromqlSeriesSample { + by: keys.into(), + kind, + child: Rc::new(base), + }) + } + Outer::TopK { k, descending } => { + // Preserve the counter-value ranking intent. Physical candidates + // may rebuild a heap over finalized rates or use exact Sort/Limit; + // neither is allowed to sum raw counter samples as ranking weights. + if descending && matches!(inner.func, Some(InnerFunc::Rate | InnerFunc::Increase)) { + let intent = inner_intent(inner.func.as_ref().expect("counter function")); + let ranked = windowed_aggregate(inner, vec![], intent); + return Ok(Unresolved::Aggregate { + reduction: Reduction::Reduce(keys.into()), + measures: vec![AggIntent::TopK { + k: k as usize, + accuracy: current_accuracy(), + }], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::new(ranked), + }); + } + // Heavy-hitter only when ranking by an additive measure (`count` + // or `sum`): that is a + // first-class aggregate intent → `TopK`. Any other ranking (topk + // over avg/quantile, a bare selector's raw value, all bottomk) + // is a generic order-by-value + limit and stays as the `Sort + Limit` + // operator pair. The descending-plus-measure rule is shared with the + // canonicalize-pass promotion so the two cannot drift (issue #38). + let measure = match inner.func { + Some(InnerFunc::Count) => topk::Ranking::Frequency, + Some(InnerFunc::Sum) => topk::Ranking::WeightedSum, + _ => topk::Ranking::NonAdditive, + }; + let additive_ranking = measure.is_supported(descending); + if additive_ranking { + // Preserve the ranked aggregate intent in the canonical tree so the + // intent algebra is explicit about what is being computed. + // Post-ASAP binding may fuse the Count and TopK into a + // single-pass heavy-hitter sketch (SpaceSaving / + // CMS-with-heap), but that is a cost-model decision, not a + // canonical-IR concern. + let ranked = match measure { + topk::Ranking::Frequency => InnerFunc::Count, + topk::Ranking::WeightedSum => InnerFunc::Sum, + topk::Ranking::NonAdditive => { + unreachable!("heavy-hitter gate rejected non-additive ranking") + } + }; + let ranked_agg = windowed_aggregate(inner, vec![], inner_intent(&ranked)); + Ok(Unresolved::Aggregate { + // A ranking always reduces (a `by`-empty TopK ranks the + // whole input into one ordering, never per-entity). + reduction: Reduction::Reduce(keys.into()), + measures: vec![AggIntent::TopK { + k: k as usize, + accuracy: current_accuracy(), + }], + output_names: vec![], + filters: vec![], + having: None, + child: Rc::new(ranked_agg), + }) + } else { + // The base over which we rank. A range-vector-function argument + // (`topk(k, rate(m[5m]))`) reduces *per series* first — that is + // label-preserving, so the `by (host)` partition labels survive. + // A **bare instant selector** (`topk(k, m)`) ranks its own + // samples directly: it must NOT be wrapped in a reducing + // aggregate. Defaulting it to `Sum` was both semantically wrong + // (PromQL `topk` ranks the raw samples, it does not sum them) and + // destructive — the cross-series `Sum` collapses every label, + // including the `by (…)` partition keys, so they no longer + // resolve (issue #30). Keep the selector label-preserving so + // `Sort.partition_by` can rank within each group (issue #12). + let base = match inner.func.as_ref().map(inner_intent) { + Some(intent) => windowed_aggregate(inner, vec![], intent), + None => instant_source(inner.metric, inner.matchers, inner.shift), + }; + Ok(ranked_by_value(keys, k, descending, base)) + } + } + } +} + +/// Decide `PerEntity` vs `Reduce(by)` for a canonical `Aggregate`, entirely +/// from local PromQL semantics: the keys and whether this operation preserves +/// each input series. It never infers entity reduction from the child tree's +/// temporal shape. `without()` is applied +/// separately, post-hoc, by `mark_without` — see its doc for why that's still +/// correct here. +fn reduction_for(keys: &[ColumnRef], per_entity: bool) -> Reduction { + if keys.is_empty() && per_entity { + Reduction::PerEntity + } else { + Reduction::Reduce(GroupKeys::by(keys.to_vec())) + } +} + +/// `Aggregate{reduction, [intent]}` over `[TimeRange{w}] → Scan`. Always wraps +/// in `TimeRange` when there's a window — including for `Rate`/`Increase`, +/// whose window rides on `inner.window` too (set redundantly alongside the +/// intent itself): canonical `AggIntent::Rate`/`Increase` carry no window +/// field of their own, unlike the old Unresolved `AggFunc::Rate{window}` — "the range +/// is on the enclosing `TimeRange` node" is now true unconditionally, so +/// there's no more `skip_window` special case. +fn windowed_aggregate( + inner: Inner, + keys: Vec, + intent: AggIntent, +) -> Unresolved { + let base = filtered_source(inner.metric, inner.matchers, inner.shift); + let child = match inner.window { + Some(w) => Unresolved::TimeRange { + range: w, + kind: TimeRangeKind::Range, + child: Rc::new(base), + }, + None => ingestion_lookback(base), + }; + let reduction = reduction_for(&keys, inner.window.is_some() || intent.is_per_series()); + Unresolved::Aggregate { + reduction, + measures: vec![intent], + // A single empty entry — never an override — so the resolver keeps + // PromQL's intent-keyed output names ("sum", "quantile_0_99", …) + // instead. + output_names: vec![String::new()], + filters: vec![], + having: None, + child: Rc::new(child), + } +} + +/// `Aggregate{reduction, [intent]}` directly over an existing Unresolved sub-DAG — the +/// OUTER level of a two-level aggregation such as `sum(rate(…))` or the +/// `Aggregate{[Quantile]}` that wraps a `histogram_quantile` argument. +fn outer_aggregate( + keys: Vec, + intent: AggIntent, + child: Unresolved, +) -> Unresolved { + let reduction = reduction_for(&keys, intent.is_per_series()); + Unresolved::Aggregate { + reduction, + measures: vec![intent], + output_names: vec![String::new()], + filters: vec![], + having: None, + child: Rc::new(child), + } +} + +/// A temporal range function over a subquery consumes each series' subquery +/// samples independently. Unlike an ordinary outer aggregate, this cannot be +/// inferred from the intent: `max` is cross-series in `max(v)`, but per-series +/// in `max_over_time(v[...])`. +fn per_series_aggregate( + keys: Vec, + intent: AggIntent, + child: Unresolved, +) -> Unresolved { + let reduction = reduction_for(&keys, true); + Unresolved::Aggregate { + reduction, + measures: vec![intent], + output_names: vec![String::new()], + filters: vec![], + having: None, + child: Rc::new(child), + } +} + +fn filtered_source(metric: String, matchers: Vec, shift: TimeShift) -> Unresolved { + let scan = Unresolved::Scan { + source: Source::TimeSeries { metric }, + predicates: matchers.into_iter().map(UnresolvedPredicate).collect(), + // Usage-derived (PromQL is schemaless) — the SchemaResolver fills this in. + schema: None, + }; + if shift.is_identity() { + scan + } else { + Unresolved::TimeShift { + shift, + child: Rc::new(scan), + } + } +} + +/// An instant selector: the latest sample per series within the workload's +/// ingestion interval, so the lookback is an `Instant` `TimeRange`. +fn instant_source(metric: String, matchers: Vec, shift: TimeShift) -> Unresolved { + ingestion_lookback(filtered_source(metric, matchers, shift)) +} + +fn ingestion_lookback(child: Unresolved) -> Unresolved { + Unresolved::TimeRange { + range: current_ingestion_interval(), + kind: TimeRangeKind::Instant, + child: Rc::new(child), + } +} + +/// Count vector elements regardless of their sample values. +fn count() -> AggIntent { + AggIntent::Count { + accuracy: current_accuracy(), + } +} + +fn inner_intent(f: &InnerFunc) -> AggIntent { + match f { + InnerFunc::FrequencyL2 => AggIntent::FrequencyL2 { + col: None, + accuracy: current_accuracy(), + }, + InnerFunc::FrequencyEntropy => AggIntent::FrequencyEntropy { + col: None, + accuracy: current_accuracy(), + }, + InnerFunc::Cardinality => AggIntent::Cardinality { + cols: vec![], + accuracy: current_accuracy(), + }, + InnerFunc::Quantile(q) => AggIntent::Quantile { + col: None, + q: *q, + accuracy: current_accuracy(), + }, + InnerFunc::Avg => AggIntent::Avg { col: None }, + InnerFunc::Min => AggIntent::Min { col: None }, + InnerFunc::Max => AggIntent::Max { col: None }, + InnerFunc::Sum => AggIntent::Sum { col: None }, + InnerFunc::StdDev => AggIntent::StdDev { + col: None, + population: true, + }, + InnerFunc::Variance => AggIntent::Variance { + col: None, + population: true, + }, + InnerFunc::Count => AggIntent::Count { + accuracy: current_accuracy(), + }, + InnerFunc::Rate => AggIntent::Rate, + InnerFunc::IRate => AggIntent::IRate, + InnerFunc::Increase => AggIntent::Increase, + InnerFunc::Changes => AggIntent::Changes, + InnerFunc::Delta => AggIntent::Delta, + InnerFunc::IDelta => AggIntent::IDelta, + InnerFunc::Deriv => AggIntent::Deriv, + InnerFunc::Resets => AggIntent::Resets, + InnerFunc::PredictLinear(s) => AggIntent::PredictLinear { seconds: *s }, + InnerFunc::DoubleExp { smoothing, trend } => AggIntent::DoubleExpSmoothing { + smoothing: *smoothing, + trend: *trend, + }, + InnerFunc::LastOverTime => AggIntent::LastOverTime, + InnerFunc::FirstOverTime => AggIntent::FirstOverTime, + InnerFunc::MadOverTime => AggIntent::MadOverTime, + InnerFunc::TsOfMinOverTime => AggIntent::TsOfMinOverTime, + InnerFunc::TsOfMaxOverTime => AggIntent::TsOfMaxOverTime, + InnerFunc::TsOfFirstOverTime => AggIntent::TsOfFirstOverTime, + InnerFunc::TsOfLastOverTime => AggIntent::TsOfLastOverTime, + } +} + +fn outer_intent(o: &OuterIntent) -> AggIntent { + match o { + OuterIntent::Sum => AggIntent::Sum { col: None }, + OuterIntent::Avg => AggIntent::Avg { col: None }, + OuterIntent::Min => AggIntent::Min { col: None }, + OuterIntent::Max => AggIntent::Max { col: None }, + OuterIntent::StdDev => AggIntent::StdDev { + col: None, + population: true, + }, + OuterIntent::Variance => AggIntent::Variance { + col: None, + population: true, + }, + OuterIntent::Quantile(q) => AggIntent::Quantile { + col: None, + q: *q, + accuracy: current_accuracy(), + }, + OuterIntent::Group => AggIntent::Group, + } +} + +/// Unwrap a (possibly parenthesised) string literal — `count_values` labels and +/// `label_replace`/`label_join` arguments are all string literals, sometimes +/// wrapped in parens (`count_values((("v")), …)`). +fn expr_str(expr: &Expr) -> Result { + match expr { + Expr::StringLiteral(s) => Ok(s.val.clone()), + Expr::Paren(p) => expr_str(&p.expr), + other => Err(LoweringError::InvalidParameter(format!( + "expected a string literal, got `{other}`" + ))), + } +} + +/// A `count_values` string parameter (the synthesized label name). +fn str_param(agg: &AggregateExpr) -> Result { + match &agg.param { + Some(e) => expr_str(e), + None => Err(LoweringError::MissingArgument( + "`count_values` label parameter".into(), + )), + } +} + +/// A call's `idx`-th argument as a string literal (`label_replace`/`label_join`). +fn str_arg(call: &Call, idx: usize) -> Result { + expr_str(arg(call, idx)?) +} + +/// Resolve an aggregation's grouping modifier into a `(keys, without)` pair. +/// +/// `by(labels)` → the kept labels, `without = false`. `without(labels)` → the +/// **excluded** labels, `without = true`: the kept set (the complement) can't be +/// enumerated under an open usage-derived schema, so it is deferred to the +/// runtime and only the excluded positions are carried (issue #39). Both forms +/// canonicalise their label set (sort + dedup) so equivalent groupings lower +/// identically. PromQL labels have no table qualifier → `ColumnRef::Named`. +fn resolve_group(agg: &AggregateExpr) -> Result<(Vec, bool)> { + let canon = |labels: &[String]| -> Vec { + let mut keys = labels.to_vec(); + keys.sort(); + keys.dedup(); + keys.into_iter().map(ColumnRef::Named).collect() + }; + match &agg.modifier { + None => Ok((vec![], false)), + Some(LabelModifier::Include(ls)) => Ok((canon(&ls.labels), false)), + Some(LabelModifier::Exclude(ls)) => Ok((canon(&ls.labels), true)), + } +} + +// ── Free helpers ────────────────────────────────────────────────────────────── + +fn vs_parts(vs: &VectorSelector) -> Result<(String, Vec, TimeShift)> { + // A non-equality `__name__` matcher (`=~` / `!~` / `!=`) selects *across* + // metric names. `Source::TimeSeries { metric }` carries a single concrete + // metric name, so there is no representation for a regex/negated name + // match — reject rather than mislower it to a literal metric named after + // the pattern (issue #67). An equality `__name__` (`{__name__="up"}`) + // still names the metric below. + if let Some(m) = vs + .matchers + .matchers + .iter() + .find(|m| m.name == "__name__" && !matches!(m.op, MatchOp::Equal)) + { + return Err(LoweringError::UnsupportedFeature(format!( + "non-equality `__name__` matcher ({}{:?}) selects across metric names, \ + which has no single-metric canonical representation", + m.name, m.op + ))); + } + let metric = vs.name.clone().unwrap_or_else(|| { + vs.matchers + .matchers + .iter() + .find(|m| m.name == "__name__") + .map(|m| m.value.clone()) + .unwrap_or_default() + }); + // Label matchers are an unordered set: `{a="1",b="2"}` and `{b="2",a="1"}` + // select the same series. Canonicalise by (name, value) so equivalent + // selectors lower to identical predicates. + let mut ms: Vec<&Matcher> = vs + .matchers + .matchers + .iter() + .filter(|m| m.name != "__name__") + .collect(); + ms.sort_by(|a, b| a.name.cmp(&b.name).then_with(|| a.value.cmp(&b.value))); + let matchers = ms.into_iter().map(matcher_to_compare).collect(); + let shift = time_shift(vs.offset.as_ref(), vs.at.as_ref())?; + Ok((metric, matchers, shift)) +} + +/// Convert the parser's `offset` / `@` modifiers into a [`TimeShift`] (issue +/// #40). Offset is signed milliseconds; `@ ` (parser seconds → ms) becomes +/// an absolute anchor, `@ start()`/`@ end()` the range bounds. +fn time_shift(offset: Option<&Offset>, at: Option<&ParserAtModifier>) -> Result { + let offset_ms = match offset { + None => 0, + Some(Offset::Pos(d)) => duration_ms(*d)?, + Some(Offset::Neg(d)) => -duration_ms(*d)?, + }; + let at = match at { + None => None, + Some(ParserAtModifier::Start) => Some(AtModifier::Start), + Some(ParserAtModifier::End) => Some(AtModifier::End), + Some(ParserAtModifier::At(t)) => Some(AtModifier::Timestamp(system_time_ms(*t)?)), + }; + Ok(TimeShift { offset_ms, at }) +} + +/// A `Duration` as `i64` milliseconds, rejecting an overflow rather than +/// silently truncating a pathologically large `offset`. +fn duration_ms(d: Duration) -> Result { + i64::try_from(d.as_millis()).map_err(|_| { + LoweringError::InvalidParameter("offset duration overflows i64 milliseconds".into()) + }) +} + +/// A `SystemTime` (`@ `) as `i64` milliseconds since the Unix epoch, signed +/// so pre-epoch anchors (the parser permits them) are preserved. +fn system_time_ms(t: SystemTime) -> Result { + let ms = match t.duration_since(std::time::UNIX_EPOCH) { + Ok(d) => i64::try_from(d.as_millis()), + Err(e) => i64::try_from(e.duration().as_millis()).map(|ms| -ms), + }; + ms.map_err(|_| { + LoweringError::InvalidParameter("`@` timestamp overflows i64 milliseconds".into()) + }) +} + +fn matcher_to_compare(m: &Matcher) -> Scalar { + let op = match &m.op { + MatchOp::Equal => CompareOpKind::Eq, + MatchOp::NotEqual => CompareOpKind::Ne, + MatchOp::Re(_) => CompareOpKind::Regex, + MatchOp::NotRe(_) => CompareOpKind::NotRegex, + }; + Scalar::Compare { + left: Box::new(Scalar::Column(ColumnRef::Named(m.name.clone()))), + op, + right: Box::new(Scalar::Literal(ScalarValue::Utf8(m.value.clone()))), + semantics: PROMQL, + } +} + +fn extract_matrix(expr: &Expr) -> Result<(String, Vec, Duration, TimeShift)> { + match expr { + Expr::MatrixSelector(ms) => { + let (metric, matchers, shift) = vs_parts(&ms.vs)?; + Ok((metric, matchers, ms.range, shift)) + } + Expr::Paren(p) => extract_matrix(&p.expr), + // A range-vector function argument must be a (parenthesised) matrix + // selector. Do NOT descend through an arbitrary `Call` — that would + // silently strip an unsupported wrapper (`rate(deriv(m[5m]))` lowering + // as `rate(m[5m])`). Reject instead. + other => Err(LoweringError::UnsupportedFeature(format!( + "expected a range-vector (matrix) argument, got `{other}`" + ))), + } +} + +fn arg(call: &Call, idx: usize) -> Result<&Expr> { + call.args + .args + .get(idx) + .map(|b| b.as_ref()) + .ok_or_else(|| LoweringError::MissingArgument(format!("{} arg #{idx}", call.func.name))) +} + +fn num_arg(call: &Call, idx: usize) -> Result { + num_expr(arg(call, idx)?) +} + +fn num_param(agg: &AggregateExpr) -> Result { + match &agg.param { + Some(e) => num_expr(e), + None => Err(LoweringError::MissingArgument( + "aggregate parameter (k / φ)".into(), + )), + } +} + +fn num_expr(expr: &Expr) -> Result { + match expr { + Expr::NumberLiteral(n) => Ok(n.val), + Expr::Paren(p) => num_expr(&p.expr), + Expr::Unary(u) => Ok(-num_expr(&u.expr)?), + // Constant-fold a pure scalar arithmetic expression — the parser does + // not fold `10*1024*1024` / `24 * 3600`. A `modifier` (vector matching) + // or a non-arithmetic operator means it is not a pure scalar. + Expr::Binary(b) if b.modifier.is_none() => { + let (l, r) = (num_expr(&b.lhs)?, num_expr(&b.rhs)?); + let id = b.op.id(); + if id == token::T_ADD { + Ok(l + r) + } else if id == token::T_SUB { + Ok(l - r) + } else if id == token::T_MUL { + Ok(l * r) + } else if id == token::T_DIV { + Ok(l / r) + } else if id == token::T_MOD { + Ok(l % r) + } else if id == token::T_POW { + Ok(l.powf(r)) + } else { + Err(LoweringError::InvalidParameter( + "non-arithmetic operator in scalar expression".into(), + )) + } + } + // `min_of`/`max_of` are n-ary *scalar* reducers (issue #89). Fold them + // when every argument is itself a constant scalar — this is the only + // form the intent algebra can hold (there is no scalar min/max node). A + // non-constant argument (`min_of(step(), 1s)`) fails the recursive fold + // and propagates the error, so it stays rejected. `f64::min`/`max` + // ignore NaN, matching PromQL's `min`/`max` NaN semantics. + Expr::Call(c) if is_scalar_reducer_fn(c.func.name) => { + let reduce = if c.func.name == "min_of" { + f64::min + } else { + f64::max + }; + c.args + .args + .iter() + .map(|a| num_expr(a)) + .reduce(|acc, v| Ok(reduce(acc?, v?))) + .ok_or_else(|| { + LoweringError::MissingArgument(format!("{} needs an argument", c.func.name)) + })? + } + other => Err(LoweringError::InvalidParameter(format!( + "expected a numeric scalar, got `{other}`" + ))), + } +} + +/// The n-ary scalar min/max reducers, foldable when all arguments are constant +/// scalars (issue #89). +fn is_scalar_reducer_fn(name: &str) -> bool { + matches!(name, "min_of" | "max_of") +} + +/// `topk`/`bottomk` count parameter — a non-negative integer. Rejects +/// fractional / negative / non-finite values rather than silently truncating +/// or saturating them via `as u64` (`topk(2.7, …)` ≠ `topk(2, …)`). +fn count_param(agg: &AggregateExpr) -> Result { + let v = num_param(agg)?; + if v.is_finite() && v >= 0.0 && v.fract() == 0.0 && v <= u64::MAX as f64 { + Ok(v as u64) + } else { + Err(LoweringError::InvalidParameter(format!( + "topk/bottomk k must be a non-negative integer, got {v}" + ))) + } +} + +/// `limit_ratio` ratio parameter — a finite value; Prometheus clamps it to +/// `[-1, 1]` (a negative ratio selects the complementary fraction). A non-finite +/// ratio (`limit_ratio(NaN, …)`) or a dynamic one (`time() % 17/17`, which +/// `num_param` can't fold) is rejected (issue #86). +fn ratio_param(agg: &AggregateExpr) -> Result { + let r = num_param(agg)?; + if !r.is_finite() { + return Err(LoweringError::InvalidParameter(format!( + "limit_ratio ratio must be finite, got {r}" + ))); + } + Ok(r.clamp(-1.0, 1.0)) +} + +/// Preserve the full Prometheus quantile parameter domain, including special values. +fn quantile_param(q: f64) -> Result { + // Prometheus returns NaN/-Inf/+Inf for these parameters at execution time. + Ok(q) +} + +// The non-standard histogram_quantiles extension keeps its bounded label contract. +fn bounded_quantile_param(q: f64) -> Result { + if q.is_finite() && (0.0..=1.0).contains(&q) { + Ok(q) + } else { + Err(LoweringError::InvalidParameter(format!( + "quantile φ must be in [0, 1], got {q}" + ))) + } +} + +fn binop(id: token::TokenId) -> Result { + Ok(if id == token::T_ADD { + BinaryOpKind::Arithmetic(ArithmeticOpKind::Add) + } else if id == token::T_SUB { + BinaryOpKind::Arithmetic(ArithmeticOpKind::Sub) + } else if id == token::T_MUL { + BinaryOpKind::Arithmetic(ArithmeticOpKind::Mul) + } else if id == token::T_DIV { + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) + } else if id == token::T_MOD { + BinaryOpKind::Arithmetic(ArithmeticOpKind::Mod) + } else if id == token::T_POW { + BinaryOpKind::Arithmetic(ArithmeticOpKind::Pow) + } else if id == token::T_ATAN2 { + BinaryOpKind::Arithmetic(ArithmeticOpKind::Atan2) + } else if id == token::T_EQLC { + BinaryOpKind::Compare(CompareOpKind::Eq) + } else if id == token::T_NEQ { + BinaryOpKind::Compare(CompareOpKind::Ne) + } else if id == token::T_LSS { + BinaryOpKind::Compare(CompareOpKind::Lt) + } else if id == token::T_LTE { + BinaryOpKind::Compare(CompareOpKind::Le) + } else if id == token::T_GTR { + BinaryOpKind::Compare(CompareOpKind::Gt) + } else if id == token::T_GTE { + BinaryOpKind::Compare(CompareOpKind::Ge) + } else if id == token::T_LAND { + BinaryOpKind::Set(PromQLVectorSetOpKind::And) + } else if id == token::T_LOR { + BinaryOpKind::Set(PromQLVectorSetOpKind::Or) + } else if id == token::T_LUNLESS { + BinaryOpKind::Set(PromQLVectorSetOpKind::Unless) + } else { + return Err(LoweringError::UnsupportedFeature(format!( + "binary operator token {id}" + ))); + }) +} diff --git a/crates/frontend-promql/tests/unified_histogram_metadata.rs b/crates/frontend-promql/tests/unified_histogram_metadata.rs new file mode 100644 index 000000000..49fdffe02 --- /dev/null +++ b/crates/frontend-promql/tests/unified_histogram_metadata.rs @@ -0,0 +1,138 @@ +//! Type-driven `histogram_quantile` discrimination (issue #79). +//! +//! The structural heuristic (`by (le)` / `_bucket` / `le=` matcher) proxies the +//! argument's sample type. A declared [`HistogramKind`] overrides it, fixing the +//! heuristic's false-positive and false-negative cases. Undeclared metrics still +//! fall back to the heuristic. + +use asap_frontend_promql::unified::{HistogramCatalog, HistogramKind}; +#[path = "unified_support.rs"] +mod support; +use asap_types::ir::{NonASAPOp, OperatorNode}; +use asap_types::pre_asap::AggIntent; +use asap_types::types::AccuracyTarget; +use support::{lower_promql, lower_promql_with_histograms}; + +/// The histogram/quantile intent kind in the lowered tree: `"HQ"` for the +/// classic-bucket `HistogramQuantile`, `"Q"` for the sketch-able `Quantile`. +fn quantile_kind(qe: &OperatorNode) -> &'static str { + fn walk(e: &OperatorNode) -> Option<&'static str> { + match e.expect_non_asap() { + NonASAPOp::Aggregate { + measures, child, .. + } => measures + .iter() + .find_map(|i| match i { + AggIntent::HistogramQuantile { .. } => Some("HQ"), + AggIntent::Quantile { .. } => Some("Q"), + _ => None, + }) + .or_else(|| walk(child)), + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } + | NonASAPOp::Project { child, .. } => walk(child), + _ => None, + } + } + walk(qe).expect("a HistogramQuantile or Quantile intent") +} + +fn heuristic(q: &str) -> &'static str { + quantile_kind(&lower_promql(q, AccuracyTarget::Exact).unwrap()) +} + +fn with_meta(q: &str, catalog: HistogramCatalog) -> &'static str { + quantile_kind(&lower_promql_with_histograms(q, AccuracyTarget::Exact, catalog).unwrap()) +} + +#[test] +fn heuristic_baseline_is_unchanged_without_a_catalog() { + // Classic buckets are represented; undeclared native samples are rejected. + assert_eq!( + heuristic( + "histogram_quantile(0.9, sum by (le) (rate(http_request_duration_seconds_bucket[5m])))" + ), + "HQ" + ); + assert!(lower_promql( + "histogram_quantile(0.9, native_latency)", + AccuracyTarget::Exact + ) + .is_err()); +} + +#[test] +fn declared_classic_bucket_fixes_the_false_negative() { + // A classic histogram exposed WITHOUT the `_bucket` suffix and queried with + // no `le` grouping/matcher requires an explicit sample-type declaration. + let q = "histogram_quantile(0.9, latency_seconds)"; + assert!(lower_promql(q, AccuracyTarget::Exact).is_err()); + assert_eq!( + with_meta( + q, + HistogramCatalog::new().with("latency_seconds", HistogramKind::ClassicBucket) + ), + "HQ", + "metadata routes it to exact bucket interpolation" + ); +} + +#[test] +fn declared_raw_extension_and_native_gap_override_the_heuristic() { + // A metric merely NAMED `…_bucket` that actually holds raw samples / a native + // histogram: the heuristic wrongly routes it to bucket interpolation. + let q = "histogram_quantile(0.9, foo_bucket)"; + assert_eq!( + heuristic(q), + "HQ", + "heuristic mis-routes on the `_bucket` name" + ); + assert_eq!( + with_meta( + q, + HistogramCatalog::new().with("foo_bucket", HistogramKind::RawSamples) + ), + "Q", + "raw samples are sketch-able" + ); + let catalog = HistogramCatalog::new().with("foo_bucket", HistogramKind::Native); + assert!(lower_promql_with_histograms(q, AccuracyTarget::Exact, catalog.clone()).is_err()); + assert!(lower_promql_with_histograms("foo_bucket", AccuracyTarget::Exact, catalog).is_err()); +} + +#[test] +fn undeclared_metric_falls_back_to_the_heuristic() { + // A catalog that doesn't mention the queried metric leaves the structural + // decision in place. + let catalog = HistogramCatalog::new().with("some_other_metric", HistogramKind::RawSamples); + assert_eq!( + with_meta( + "histogram_quantile(0.9, sum by (le) (x_bucket))", + catalog.clone() + ), + "HQ" + ); + assert!(lower_promql_with_histograms( + "histogram_quantile(0.9, native_thing)", + AccuracyTarget::Exact, + catalog + ) + .is_err()); +} + +#[test] +fn the_catalog_does_not_leak_across_calls() { + // The ambient catalog is scoped to the single `_with_histograms` call; a + // subsequent plain `lower_promql` sees no metadata (guards against a + // thread-local that isn't cleaned up). + let _ = with_meta( + "histogram_quantile(0.9, foo_bucket)", + HistogramCatalog::new().with("foo_bucket", HistogramKind::RawSamples), + ); + // `foo_bucket` would be sketch-able under that catalog, but with none it must + // revert to the heuristic (the `_bucket` name → HistogramQuantile). + assert_eq!(heuristic("histogram_quantile(0.9, foo_bucket)"), "HQ"); +} diff --git a/crates/frontend-promql/tests/unified_promql_conformance.rs b/crates/frontend-promql/tests/unified_promql_conformance.rs new file mode 100644 index 000000000..6ea89a0bb --- /dev/null +++ b/crates/frontend-promql/tests/unified_promql_conformance.rs @@ -0,0 +1,2208 @@ +//! PromQL **semantic conformance** for the parse-to-canonical-tree lowering. +//! +//! We *lower* PromQL to the intent algebra; we do not *execute* it. So "same +//! semantic job as Prometheus" here means: for each canonical query, does the +//! canonical tree encode the **documented PromQL meaning** — and where we knowingly +//! diverge (reject, approximate, or drop a modifier), is that pinned by a test +//! so it stays visible? +//! +//! Sources for the queries + their semantics: +//! - PromQL basics (data types, selectors, offset/@/subquery): +//! +//! - PromLabs PromQL cheat sheet (common real-world queries by category): +//! +//! - Prometheus' own engine test corpus (these are *execution* tests — +//! load → eval → expect values — so they define semantics we mirror as +//! *structure*): +//! Relevant files, mapped to the sections below: selectors.test, +//! aggregators.test, functions.test, histograms.test, operators.test, +//! subquery.test, at_modifier.test, literals.test, limit.test +//! +//! Legend used in test names: +//! - (no suffix) — we lower it and the canonical intent matches PromQL. +//! - `__GAP` — a PromQL capability we don't *yet* support. It is **cleanly +//! rejected** (never silently mislowered), and pinned here so adding support +//! later flips the assertion deliberately. +//! +//! NOTE: the formerly-silent divergences (`group`→sum, dropped `offset`/`@`, +//! `changes`/`resets`→count) are now rejected rather than mislowered — see the +//! equivalence suite (`promql_equivalence.rs`) and section L below. + +// `__GAP`-suffixed test names intentionally SHOUT the documented divergences. +#![allow(non_snake_case)] + +use std::rc::Rc; +use std::time::Duration; + +use asap_frontend_promql::unified::PromqlError as LoweringError; +#[path = "unified_support.rs"] +mod support; +use asap_types::ir::{ + BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, ScalarExpr, TimeRangeKind, +}; +use asap_types::pre_asap::schema::DataType; +use asap_types::pre_asap::{ + AggIntent, ArithmeticOpKind, AtModifier, BinaryOpKind, CompareOpKind, PromQLVectorSetOpKind, + Reduction, SampleKind, ScalarValue, Source, TimeFunc, +}; +use asap_types::types::AccuracyTarget; +use support::{lower_promql, promql_scalar}; + +// ── harness helpers ───────────────────────────────────────────────────────────── + +/// Lower, expecting success. +fn ok(q: &str) -> Rc { + lower_promql(q, AccuracyTarget::Exact) + .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) +} + +/// Lower, expecting a clean `LoweringError` (an unsupported capability). +fn rejected(q: &str) -> LoweringError { + match lower_promql(q, AccuracyTarget::Exact) { + Err(e) => e, + Ok(tree) => panic!("expected {q:?} to be rejected, but it lowered to: {tree:?}"), + } +} + +/// Every `AggIntent` anywhere in the tree, root-to-leaf. +fn intents(e: &OperatorNode) -> Vec { + let mut out = Vec::new(); + collect(e, &mut out); + out +} + +/// `AggIntent` only ever lives in `Aggregate.measures`, never in a scalar +/// position (issue #205); `children()` also descends into the operators a +/// scalar position reads (`scalar(v)`). +fn collect(e: &OperatorNode, out: &mut Vec) { + if let Some(NonASAPOp::Aggregate { measures, .. }) = e.non_asap() { + out.extend(measures.iter().cloned()); + } + for child in e.children() { + collect(child, out); + } +} + +/// The first `Scan` reached by descending single-child nodes, with its metric +/// name and predicate count. +fn first_scan(e: &OperatorNode) -> (String, usize) { + match e.expect_non_asap() { + NonASAPOp::Scan { + source, predicates, .. + } => { + let name = match source { + Source::TimeSeries { metric } => metric.clone(), + Source::Table { table_ref } => table_ref.clone(), + }; + (name, predicates.len()) + } + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::TimeShift { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } + | NonASAPOp::PromqlSubquery { child, .. } => first_scan(child), + other => panic!("no Scan reachable from {other:?}"), + } +} + +fn has bool>(e: &OperatorNode, pred: F) -> bool { + intents(e).iter().any(pred) +} + +/// Whether the tree contains a `Mul`-by-`ScalarExpr(-1)` anywhere — the shape unary +/// negation lowers to (issue #36). +fn negates_via_scalar(e: &OperatorNode) -> bool { + fn negative(expr: &ScalarExpr) -> bool { + matches!(expr, ScalarExpr::Negative { .. }) || expr.children().iter().any(|e| negative(e)) + } + e.expect_non_asap() + .scalar_exprs() + .iter() + .any(|e| negative(e)) + || e.children().iter().any(|e| negates_via_scalar(e)) +} + +// ───────────────────────────────────────────────────────────────────────────── +// A. Selectors & label matchers (basics §"Instant/Range Vector +// Selectors"; selectors.test) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn instant_vector_selector() { + // SEMANTICS: bare metric → instant vector (latest sample per series). + let (metric, preds) = first_scan(&ok("node_cpu_seconds_total")); + assert_eq!(metric, "node_cpu_seconds_total"); + assert_eq!(preds, 0, "no label matchers → no predicates"); +} + +#[test] +fn promql_scan_schema_is_open() { + // A schemaless PromQL leaf is *open*: the metric's full label set is + // runtime-only, so the binding schema lists only the (ts, value) floor + + // referenced labels and may be a subset of the runtime row. + let qe = ok("node_cpu_seconds_total"); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { + panic!("expected a TimeRange for a bare selector, got {qe:?}"); + }; + let NonASAPOp::Scan { schema, .. } = child.expect_non_asap() else { + panic!("expected a Scan inside the TimeRange, got {qe:?}"); + }; + assert!( + !schema.closed, + "a schemaless PromQL scan has an open schema" + ); +} + +#[test] +fn label_matchers_become_scan_predicates() { + // SEMANTICS: `=`, `!=`, `=~`, `!~` filter series; one conjunct per matcher. + let (_, preds) = first_scan(&ok( + r#"http_requests_total{job!="x",path=~"/api/.*",env!~"dev"}"#, + )); + assert_eq!(preds, 3, "three matchers → three Scan predicates"); +} + +#[test] +fn name_label_selects_the_metric() { + // SEMANTICS: the metric name is the internal `__name__` label. + let (metric, preds) = first_scan(&ok(r#"{__name__="up"}"#)); + assert_eq!(metric, "up"); + assert_eq!(preds, 0, "__name__ is the metric, not a residual predicate"); +} + +#[test] +fn name_regex_matcher_is_rejected__GAP() { + // A `__name__=~` / `!~` / `!=` matcher selects *across* metric names, which + // the single-metric `Source::TimeSeries { metric }` can't represent. It is + // rejected (issue #67) rather than silently mislowered to a literal metric + // named after the pattern (`{__name__=~"node_.*"}` → `Source("node_.*")`). + // Full support needs a wildcard/regex `Source` in the IR. + let _ = rejected(r#"{__name__=~"node_.*"}"#); + let _ = rejected(r#"{__name__!~"x", job="y"}"#); + // Equality still names the metric (regression guard for the fix). + let (metric, _) = first_scan(&ok(r#"{__name__="up"}"#)); + assert_eq!(metric, "up"); +} + +#[test] +fn range_vector_selector_is_time_range() { + // SEMANTICS: `[5m]` turns an instant vector into a range vector, + // represented in the canonical tree as a dedicated `TimeRange` node. + let qe = ok("node_cpu_seconds_total[5m]"); + let NonASAPOp::TimeRange { range, .. } = qe.expect_non_asap() else { + panic!("expected TimeRange for a range-vector selector, got {qe:?}"); + }; + assert_eq!(*range, Duration::from_secs(300)); +} + +// ───────────────────────────────────────────────────────────────────────────── +// B. Counters: rate / irate / increase (cheat sheet "Rates of Increase"; +// functions.test) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn selector_time_ranges_carry_their_kind() { + // SEMANTICS: an instant selector reads the latest sample within the + // ingestion interval (`Instant`); `m[5m]` is a range selection (`Range`). + // Same length is not the same shape: `m` and `m[1s]` stay distinct. + assert!(matches!( + ok("node_cpu_seconds_total").expect_non_asap(), + NonASAPOp::TimeRange { + kind: TimeRangeKind::Instant, + .. + } + )); + assert!(matches!( + ok("node_cpu_seconds_total[5m]").expect_non_asap(), + NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, + .. + } + )); + assert_ne!( + ok("node_cpu_seconds_total"), + ok("node_cpu_seconds_total[1s]") + ); +} + +#[test] +fn rate_range_lives_in_time_range_node() { + // SEMANTICS: per-second average rate; the temporal range lives on the + // enclosing `TimeRange` node, not inside the intent. + let qe = ok("rate(http_requests_total[5m])"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected Aggregate, got {qe:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Rate])); + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { + panic!("expected TimeRange child, got {child:?}"); + }; + assert_eq!(*range, Duration::from_secs(300)); +} + +#[test] +fn irate_maps_to_its_own_intent() { + assert!(has(&ok("irate(http_requests_total[1m])"), |i| matches!( + i, + AggIntent::IRate + ))); +} + +#[test] +fn increase_range_lives_in_time_range_node() { + let qe = ok("increase(http_requests_total[1h])"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected Aggregate, got {qe:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Increase])); + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { + panic!("expected TimeRange child, got {child:?}"); + }; + assert_eq!(*range, Duration::from_secs(3600)); +} + +// ───────────────────────────────────────────────────────────────────────────── +// C. Aggregation across series (cheat sheet "Aggregating Over +// Multiple Series"; aggregators.test) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn sum_collapses_all_series() { + // SEMANTICS: `sum(v)` → one output series. No grouping → no Partition. + let qe = ok("sum(node_filesystem_size_bytes)"); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. })); + assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); +} + +#[test] +fn sum_by_groups_via_positional_aggregate() { + // SEMANTICS: `by(job,instance)` keeps those labels; the grouping lives on a + // positional `Aggregate.by` — the same shape SQL `GROUP BY` produces (not a + // name-based Partition). SchemaResolver leaf = [ts, value, instance, job] (referenced + // keys appended sorted), so the keys resolve to columns [2, 3]. + let qe = ok("sum by(job, instance) (node_filesystem_size_bytes)"); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected positional Aggregate for `by(...)`, got {qe:?}"); + }; + assert_eq!( + reduction, + &Reduction::by(vec![2, 3]), + "group keys resolve to positional ColumnIds" + ); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + assert!( + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) + ); +} + +#[test] +fn count_is_row_count() { + assert!(has(&ok("count(up)"), |i| matches!( + i, + AggIntent::Count { .. } + ))); +} + +#[test] +fn avg_min_max_stddev_stdvar_quantile_aggregators() { + assert!(has(&ok("avg(up)"), |i| matches!(i, AggIntent::Avg { .. }))); + assert!(has(&ok("min(up)"), |i| matches!(i, AggIntent::Min { .. }))); + assert!(has(&ok("max(up)"), |i| matches!(i, AggIntent::Max { .. }))); + assert!(has(&ok("stddev(up)"), |i| matches!( + i, + AggIntent::StdDev { .. } + ))); + assert!(has(&ok("stdvar(up)"), |i| matches!( + i, + AggIntent::Variance { .. } + ))); + assert!(has(&ok("quantile(0.5, up)"), |i| matches!( + i, + AggIntent::Quantile { .. } + ))); +} + +#[test] +fn sum_without_groups_by_the_complement() { + // SEMANTICS (issue #39): `without(instance)` = group by all labels EXCEPT + // instance. The complement can't be enumerated under the open usage-derived + // schema, so the excluded label is stored and the kept set is deferred to + // the runtime: the grouping is the exclusion form and the output schema + // stays OPEN (unlike `by`, which freezes to closed). + let qe = ok("sum without(instance) (node_filesystem_size_bytes)"); + let NonASAPOp::Aggregate { + reduction, + measures, + .. + } = qe.expect_non_asap() + else { + panic!("expected an Aggregate, got {qe:?}"); + }; + let by = reduction.expect_reduce(); + assert!( + by.is_without(), + "the grouping is the `without` exclusion form" + ); + assert_eq!(by.keys().len(), 1, "the one excluded label (instance)"); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + assert!( + !qe.schema.clone().closed, + "a `without` result keeps an open schema (kept label set is runtime-only)" + ); +} + +#[test] +fn without_on_topk_is_rejected() { + // `without(...)` is modelled only for reducing aggregations; on topk/bottomk + // (a ranking, not a reduction) it would need without-partitioning, so it is + // rejected rather than silently lowered as a `by` (issue #39). + let e = rejected("topk without (job) (3, http_requests_total)"); + assert!(format!("{e}").contains("without"), "got {e}"); +} + +#[test] +fn group_aggregator_lowers_to_a_distinct_intent() { + // SEMANTICS (PromQL): `group(v)` returns a constant 1 per group (presence), + // NOT a sum. It now lowers to a distinct `Group` intent (never folded onto + // `Sum`) — see §S. Regression guard that it is not a `Sum`. + let qe = ok("group by (job) (up)"); + assert!(has(&qe, |i| *i == AggIntent::Group)); + assert!(!has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); +} + +// ───────────────────────────────────────────────────────────────────────────── +// D. Two-level: outer aggregation OVER an inner counter (the canonical +// `sum(rate(...))` shape; aggregators.test + functions.test) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn sum_of_rate_is_two_levels() { + // SEMANTICS: per-series rate, THEN cross-series sum. Both must survive. + let qe = ok("sum(rate(http_requests_total[5m]))"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate{{Sum}}, got {qe:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + )); +} + +#[test] +fn sum_by_of_rate_groups_outer_level() { + // Outer cross-series Sum grouped on positional `Aggregate.by` over the + // label-preserving inner Rate. Leaf = [ts, value, instance] → by = [2]. + let qe = ok("sum by(instance) (rate(node_network_receive_bytes_total[5m]))"); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate grouped by instance, got {qe:?}"); + }; + assert_eq!(reduction, &Reduction::by(vec![2])); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + // child is the inner per-series Rate aggregate. + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + )); +} + +#[test] +fn sum_by_of_over_time_groups_outer_level() { + // Outer cross-series Sum grouped on positional `Aggregate.by` over an inner + // *per-series* `avg_over_time` — `Window { Aggregate{Avg} }` is label- + // preserving, so the key resolves positionally just like the rate case (no + // name-based Partition). Leaf = [ts, value, instance] → by = [2]. + let qe = ok("sum by(instance) (avg_over_time(node_cpu_seconds_total[5m]))"); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate grouped by instance, got {qe:?}"); + }; + assert_eq!(reduction, &Reduction::by(vec![2])); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + // child is the inner per-series reduction: Aggregate{Avg} over TimeRange. + let NonASAPOp::Aggregate { + measures, child, .. + } = child.expect_non_asap() + else { + panic!("expected Aggregate (per-series avg_over_time) under the Sum, got {child:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); +} + +// ───────────────────────────────────────────────────────────────────────────── +// E. Aggregation over time (per-series) (cheat sheet "Aggregating Over +// Time"; functions.test) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn over_time_functions_reduce_over_time_range() { + // SEMANTICS: reduce the samples WITHIN each series over the range → + // Aggregate over TimeRange (per-series, label-preserving). + for (q, want) in [ + ("avg_over_time(go_goroutines[5m])", "avg"), + ("max_over_time(process_resident_memory_bytes[1d])", "max"), + ("min_over_time(go_goroutines[5m])", "min"), + ("sum_over_time(go_goroutines[5m])", "sum"), + ("count_over_time(go_goroutines[5m])", "count"), + ] { + let qe = ok(q); + assert!( + matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. }), + "{q}: expected Aggregate" + ); + let matched = intents(&qe).iter().any(|i| match want { + "avg" => matches!(i, AggIntent::Avg { .. }), + "max" => matches!(i, AggIntent::Max { .. }), + "min" => matches!(i, AggIntent::Min { .. }), + "sum" => matches!(i, AggIntent::Sum { .. }), + "count" => matches!(i, AggIntent::Count { .. }), + _ => unreachable!(), + }); + assert!(matched, "{q}: missing {want} intent"); + } +} + +#[test] +fn quantile_over_time_is_aggregate_over_time_range() { + let qe = ok("quantile_over_time(0.9, request_latency_seconds[5m])"); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. })); + assert!(has( + &qe, + |i| matches!(i, AggIntent::Quantile { q, .. } if (*q - 0.9).abs() < 1e-9) + )); +} + +// ───────────────────────────────────────────────────────────────────────────── +// F. Histograms (cheat sheet "Quantiles from +// Histograms"; histograms.test) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn histogram_quantile_over_rate() { + // φ-quantile from bucket rates. The `_bucket` metric marks the classic + // cumulative-bucket form → `HistogramQuantile` (even without `sum by (le)`). + let qe = ok("histogram_quantile(0.9, rate(demo_api_request_duration_seconds_bucket[5m]))"); + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { + panic!("expected Aggregate{{HistogramQuantile}}, got {qe:?}"); + }; + assert!( + matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if (*q - 0.9).abs() < 1e-9) + ); + assert!(has(&qe, |i| matches!(i, AggIntent::Rate))); +} + +#[test] +fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { + // SEMANTICS: the standard pattern — bucket rates summed by `le`, then the + // quantile. The `sum by (le)` aggregation must survive into the + // canonical tree. + let qe = ok( + "histogram_quantile(0.99, sum by(le) (rate(demo_api_request_duration_seconds_bucket[5m])))", + ); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate{{HistogramQuantile}}, got {qe:?}"); + }; + // `by (le)` marks the classic cumulative-bucket form → `HistogramQuantile`. + assert!(matches!( + measures.as_slice(), + [AggIntent::HistogramQuantile { .. }] + )); + // `sum by(le)` now survives as a positional Aggregate (by = [2], `le`), over + // the inner Rate — no name-based Partition. + let NonASAPOp::Aggregate { + reduction, + measures, + .. + } = child.expect_non_asap() + else { + panic!("expected `sum by(le)` as a positional Aggregate, got {child:?}"); + }; + assert_eq!(reduction, &Reduction::by(vec![2])); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); +} + +// ───────────────────────────────────────────────────────────────────────────── +// G. Binary ops: math, matching, comparison (cheat sheet "Math Between +// Series" / "Filtering Series by Value"; operators.test) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn vector_arithmetic() { + let qe = ok("node_memory_MemFree_bytes + node_memory_Cached_bytes"); + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + .. + } = qe.expect_non_asap() + else { + panic!("expected BinaryOp, got {qe:?}"); + }; + assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Add)); +} + +#[test] +fn on_matching_with_group_left() { + // SEMANTICS: many-to-one matching on a label subset. + let qe = + ok("rate(demo_cpu_usage_seconds_total[1m]) / on(instance, job) group_left demo_num_cpus"); + let NonASAPOp::BinaryOp { operator, .. } = qe.expect_non_asap() else { + panic!("expected BinaryOp, got {qe:?}"); + }; + assert_eq!( + operator.kind, + BinaryOpKind::Arithmetic(ArithmeticOpKind::Div) + ); + let vm = operator + .vector_match + .as_ref() + .expect("on(...) group_left present"); + assert_eq!(vm.labels, vec!["instance".to_string(), "job".to_string()]); + assert!( + vm.grouping.is_some(), + "group_left should set the grouping side" + ); +} + +#[test] +fn vector_comparison_filters() { + // SEMANTICS: `>` between two vectors keeps the LHS series where it holds. + let qe = ok("go_goroutines > go_threads"); + assert!( + matches!(qe.expect_non_asap(), NonASAPOp::BinaryOp { operator: BinaryOperator { kind: op, .. }, .. } if *op == BinaryOpKind::Compare(CompareOpKind::Gt)) + ); +} + +#[test] +fn comparison_bool_modifier_returns_zero_or_one() { + // SEMANTICS (operators.test): `bool` turns a filtering comparison into a + // 0/1-valued one. On a vector operand it is `return_bool` on the + // `BinaryOp`; between two scalars it is a `Case(Compare → 1, else 0)` + // scalar expression under PromQL numeric rules — and a scalar comparison + // without `bool` is not a PromQL expression at all. + let bool_flag = |q: &str| match ok(q).expect_non_asap() { + NonASAPOp::BinaryOp { return_bool, .. } => *return_bool, + NonASAPOp::Project { .. } => true, + NonASAPOp::Filter { .. } => false, + other => panic!("expected BinaryOp for {q}, got {other:?}"), + }; + assert!(bool_flag("go_goroutines > bool go_threads")); + assert!(bool_flag("go_goroutines > bool 0")); + assert!(!bool_flag("go_goroutines > go_threads")); + assert!(!bool_flag("go_goroutines > 0")); + + let qe = support::scalar_root("1 < bool 2"); + let ScalarExpr::Case { branches, .. } = &qe else { + panic!("expected a scalar Case, got {qe:?}"); + }; + assert!(matches!( + branches.as_slice(), + [( + ScalarExpr::Compare { + op: CompareOpKind::Lt, + semantics: ExprSemantics::Promql, + .. + }, + _ + )] + )); + rejected("1 < 2"); +} + +#[test] +fn unary_negation_lowers_as_multiply_by_minus_one() { + // SEMANTICS (PromQL, issue #36): `-expr` flips the sign of every sample. + // Now that a scalar operand exists (#35), it lowers as `expr * -1` — a `Mul` + // BinaryOp of the (label-preserving) vector against `ScalarExpr(-1)`. These are + // the five cases the old `__GAP` test pinned as rejected. + for q in [ + "-rate(http_errors_total[5m])", + "-some_metric", + "-metric_a or -metric_b", + "http_requests_total - -http_errors_total", + "sum(-node_cpu_seconds_total)", + ] { + let qe = ok(q); + // A `Mul`-by-`-1` against a `ScalarExpr(-1)` appears somewhere in every tree. + assert!( + negates_via_scalar(&qe), + "no `* -1` negation found in {q}: {qe:?}" + ); + } + + let negated = ok("-some_metric"); + assert!(negates_via_scalar(&negated)); + assert!(negated.schema.has_promql_series_identity()); + assert!(negated.schema.time_index.is_some()); + let summed = ok("sum(-node_cpu_seconds_total)"); + assert!(has(&summed, |i| matches!(i, AggIntent::Sum { .. }))); + assert!(negates_via_scalar(&summed)); +} + +#[test] +fn unary_negation_of_constant_folds_to_scalar() { + // `-(10*1024*1024)` — the operand is constant-foldable, so negation collapses + // to a single negated `ScalarExpr` leaf (no `BinaryOp`), just like a bare literal. + assert!(promql_scalar(&support::scalar_root("-(10*1024*1024)")) + .is_some_and(|v| (v + 10_485_760.0).abs() < 1e-6)); +} + +#[test] +fn double_unary_negation_nests() { + let qe = ok("- -some_metric"); + let NonASAPOp::Project { child, .. } = qe.expect_non_asap() else { + panic!() + }; + assert!(matches!(child.expect_non_asap(), NonASAPOp::Project { .. })); + assert!(negates_via_scalar(child)); +} + +#[test] +fn count_maps_to_count_and_inherits_accuracy() { + // Counts preserve the workload accuracy target without counting distinct values. + let exact = lower_promql("count by (job) (up)", AccuracyTarget::Exact).unwrap(); + assert!( + has(&exact, |i| matches!( + i, + AggIntent::Count { + accuracy: AccuracyTarget::Exact + } + )), + "Count must stay Exact under AccuracyTarget::Exact, got {:?}", + intents(&exact) + ); + + let approx = lower_promql("count by (job) (up)", AccuracyTarget::Epsilon(0.01)).unwrap(); + assert!( + has(&approx, |i| matches!( + i, + AggIntent::Count { + accuracy: AccuracyTarget::Epsilon(e) + } if (*e - 0.01).abs() < 1e-9 + )), + "Count must carry the approximate target, got {:?}", + intents(&approx) + ); +} + +#[test] +fn scalar_literal_operand_lowers_as_binaryop_scalar() { + let qe = ok("node_filesystem_avail_bytes > 10*1024*1024"); + let ScalarExpr::Compare { op, right, .. } = support::sample_expression(&qe) else { + panic!() + }; + assert_eq!(*op, CompareOpKind::Gt); + assert_eq!(promql_scalar(right), Some(10_485_760.0)); +} + +#[test] +fn scalar_arithmetic_scales_the_vector() { + let qe = ok("rate(m[5m]) * 100"); + let ScalarExpr::Arithmetic { op, right, .. } = support::sample_expression(&qe) else { + panic!() + }; + assert_eq!(*op, ArithmeticOpKind::Mul); + assert_eq!(promql_scalar(right), Some(100.0)); +} + +// ───────────────────────────────────────────────────────────────────────────── +// H. Set operations (cheat sheet "Set Operations"; +// operators.test) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn set_ops_lower_to_binaryop() { + // SEMANTICS: or = union of label sets; and = intersection; unless = difference. + let set_op = |q: &str| match ok(q).expect_non_asap() { + NonASAPOp::BinaryOp { operator, .. } => operator.kind.clone(), + other => panic!("expected BinaryOp for {q}, got {other:?}"), + }; + assert_eq!( + set_op("up{job=\"a\"} or up{job=\"b\"}"), + BinaryOpKind::Set(PromQLVectorSetOpKind::Or) + ); + assert_eq!( + set_op("node_network_mtu_bytes and node_up"), + BinaryOpKind::Set(PromQLVectorSetOpKind::And) + ); + assert_eq!( + set_op("node_network_mtu_bytes unless node_down"), + BinaryOpKind::Set(PromQLVectorSetOpKind::Unless) + ); +} + +// ───────────────────────────────────────────────────────────────────────────── +// I. Sorting / top-k (cheat sheet "Sorting"/topk; +// functions.test, limit.test) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn topk_over_count_is_heavy_hitter() { + // SEMANTICS: top-k by frequency → first-class heavy-hitter `TopK` intent. + let qe = ok("topk(10, count_over_time(http_requests_total[1m]))"); + assert!(has( + &qe, + |i| matches!(i, AggIntent::TopK { k, .. } if *k == 10) + )); +} + +#[test] +fn bottomk_is_generic_sort_limit() { + // SEMANTICS: bottom-k → generic ascending order + limit (no sketch). + let qe = ok("bottomk(3, count_over_time(http_requests_total[5m]))"); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Limit { .. })); +} + +#[test] +fn topk_over_nested_sum_preserves_weighted_topk_accuracy() { + // SEMANTICS (PromQL): `topk(3, sum by(x)(rate(...)))` is extremely common. + // The final rates are query-time values. Their ordering does not establish + // frequency-sketch membership semantics. + let qe = ok("topk(3, sum by(instance) (rate(node_cpu_seconds_total[5m])))"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected weighted TopK aggregate, got {qe:?}"); + }; + assert!(matches!( + measures.as_slice(), + [AggIntent::TopK { k: 3, .. }] + )); + // The inner `sum by (instance)` survives as a cross-series Aggregate over the + // per-series rate — the nesting the old two-level template could not express. + assert!( + has(child, |i| matches!(i, AggIntent::Sum { .. })) + && has(child, |i| matches!(i, AggIntent::Rate)), + "inner sum-over-rate preserved, got {:?}", + intents(child) + ); + assert!(has(&qe, |i| matches!(i, AggIntent::TopK { .. }))); +} + +#[test] +fn outer_aggregate_over_nested_aggregate_nests() { + // `max(sum by (job) (rate(m[5m])))` — an outer cross-series reduction over a + // nested per-group reduction over a per-series rate: three stacked levels the + // flat two-level template rejected. Each level survives into the + // canonical tree (issue #27). + let qe = ok("max(sum by (job) (rate(http_requests_total[5m])))"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate, got {qe:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); + let NonASAPOp::Aggregate { + reduction, + measures, + .. + } = child.expect_non_asap() + else { + panic!("expected inner `sum by (job)` Aggregate, got {child:?}"); + }; + assert_eq!( + reduction, + &Reduction::by(vec![2]), + "job grouping survives on the inner aggregate" + ); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + assert!(has(&qe, |i| matches!(i, AggIntent::Rate)), "rate preserved"); +} + +#[test] +fn outer_group_key_absent_from_nested_aggregate_is_dropped() { + // SEMANTICS (PromQL, issue #53): aggregating `by` a label that no input + // series carries is valid — every series lands in one group and the + // (empty) label is omitted from the output. Here the inner `sum by (group)` + // collapses `job` away (its closed output schema is `[group, sum]`), so the + // outer `by (job)` groups everything into a single global partition: + // the query lowers with the provably-absent key dropped, exactly + // `sum(sum by (group)(…))`. + let qe = ok(r#"sum(sum by (group)(http_requests{job="api-server"})) by (job)"#); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate, got {qe:?}"); + }; + assert_eq!( + reduction, + &Reduction::by(vec![]), + "absent `job` key dropped → global aggregate" + ); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + let NonASAPOp::Aggregate { reduction, .. } = child.expect_non_asap() else { + panic!("expected inner `sum by (group)` Aggregate, got {child:?}"); + }; + assert_eq!( + reduction, + &Reduction::by(vec![2]), + "inner grouping on `group` survives" + ); +} + +#[test] +fn outer_group_key_present_after_inner_aggregate_still_resolves() { + // The counterpart guard for #53: when the outer key IS in the inner + // aggregate's output (`by (job)` over `sum by (job, group)`), it must keep + // resolving positionally — the absent-key drop only fires on provable + // absence, never on a resolvable key. + let qe = ok("sum(sum by (job, group)(http_requests)) by (job)"); + let NonASAPOp::Aggregate { + reduction, child, .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate, got {qe:?}"); + }; + let NonASAPOp::Aggregate { + reduction: inner_reduction, + .. + } = child.expect_non_asap() + else { + panic!("expected inner Aggregate, got {child:?}"); + }; + // Inner output schema is [group, job, sum] (keys in label-column order, + // labels alphabetical on the scan) → job = col 1. + assert_eq!( + reduction, + &Reduction::by(vec![1]), + "outer `job` resolves against the inner output" + ); + assert_eq!(inner_reduction.expect_reduce().len(), 2); +} + +#[test] +fn outer_group_key_over_binary_op_resolves_on_both_sides() { + // Issue #52: an outer aggregate's group key that appears in *neither* side of + // a binary op — the metric-name label `__name__`, or a plain `job` — must + // still resolve. Each `or` side is bound independently against its own + // sub-tree, so the key is seeded as an inherited column on both sides. + let qe = ok(r#"sum by (__name__)(metric_a{env="1"} or metric_b{env="2"})"#); + let NonASAPOp::Aggregate { + reduction, child, .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate, got {qe:?}"); + }; + // `__name__` resolves to a single positional id against the binary op output. + assert_eq!( + reduction.expect_reduce().len(), + 1, + "grouped by the one `__name__` key" + ); + let NonASAPOp::BinaryOp { lhs, rhs, .. } = child.expect_non_asap() else { + panic!("expected a BinaryOp child, got {child:?}"); + }; + // Both independently-bound sides carry `__name__` at the same position, so + // the outer group key is consistent across the union. + let (ls, rs) = (lhs.schema.clone(), rhs.schema.clone()); + assert_eq!(ls.column_id("__name__"), rs.column_id("__name__")); + assert_eq!( + ls.column_id("__name__"), + Some(reduction.expect_reduce().keys()[0]) + ); + + // The general case (a plain label, not just `__name__`) also lowers. + assert!(matches!( + ok("sum by (job)(metric_a or metric_b)").expect_non_asap(), + NonASAPOp::Aggregate { .. } + )); +} + +#[test] +fn aggregate_over_binary_op_nests() { + // `sum(rate(a[5m]) + rate(b[5m]))` — an aggregate whose argument is a binary + // op over two range vectors. The old template only accepted a single inner + // selector/call; now the binary op lowers and the outer sum wraps it. + let qe = ok("sum(rate(a[5m]) + rate(b[5m]))"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate, got {qe:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + assert!( + matches!(child.expect_non_asap(), NonASAPOp::BinaryOp { .. }), + "argument lowers as a BinaryOp, got {child:?}" + ); +} + +// ───────────────────────────────────────────────────────────────────────────── +// J. Subqueries (basics §Subqueries; subquery.test) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn subquery_wraps_inner_query() { + // SEMANTICS: `[range:res]` evaluates the inner query across a range. + let qe = ok("rate(demo_api_request_duration_seconds_count[5m])[1h:]"); + assert!(matches!( + qe.expect_non_asap(), + NonASAPOp::PromqlSubquery { .. } + )); + assert!(has(&qe, |i| matches!(i, AggIntent::Rate))); +} + +#[test] +fn over_time_of_subquery_reduces_per_series() { + // SEMANTICS (PromQL): `max_over_time(rate(...)[1h:])` chains a sub-query into + // a range-vector function — the sub-query evaluates `rate` across a 1h range, + // then `max_over_time` takes the max of those samples *per series*. It lowers + // to a per-series `Max` reduction over a `PromqlSubquery` (issue #27). + let qe = ok("max_over_time(rate(demo_api_request_duration_seconds_count[5m])[1h:])"); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected an Aggregate at the root, got {qe:?}"); + }; + assert_eq!( + reduction, + &Reduction::PerEntity, + "`*_over_time` has no grouping — reduces per series" + ); + assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); + // The reduction rides directly on the sub-query (the structural range marker + // that keeps it label-preserving), which wraps the inner `rate`. + assert!( + matches!(child.expect_non_asap(), NonASAPOp::PromqlSubquery { .. }), + "the `Max` reduces over a PromqlSubquery, got {child:?}" + ); + assert!(intents(&qe).iter().any(|i| matches!(i, AggIntent::Rate))); +} + +#[test] +fn quantile_over_time_of_subquery_carries_phi() { + // The `quantile_over_time` φ parameter is read from arg 0; the sub-query is + // arg 1. It lowers to a per-series `Quantile(φ)` over the `PromqlSubquery`. + let qe = ok("quantile_over_time(0.9, rate(demo[5m])[1h:])"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected an Aggregate, got {qe:?}"); + }; + assert!( + matches!(measures.as_slice(), [AggIntent::Quantile { q, .. }] if (*q - 0.9).abs() < 1e-9) + ); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::PromqlSubquery { .. } + )); +} + +#[test] +fn aggregation_over_over_time_of_subquery_keeps_labels() { + // `sum by (job) (max_over_time(rate(m[5m])[1h:]))` — the inner + // `max_over_time` is per-series (label-preserving), so the `job` label + // survives for the OUTER cross-series `sum by (job)` to group on. If the + // inner `Max` collapsed labels, `job` would not resolve here. + let qe = ok("sum by (job) (max_over_time(rate(demo{job=\"api\"}[5m])[1h:]))"); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate, got {qe:?}"); + }; + assert!( + matches!(reduction, Reduction::Reduce(by) if !by.is_empty()), + "outer `sum by (job)` groups on a label" + ); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + // Inner node is the per-series `max_over_time` reduction over the subquery. + let NonASAPOp::Aggregate { + reduction: inner_reduction, + measures: inner_measures, + child: inner_child, + .. + } = child.expect_non_asap() + else { + panic!("expected inner Aggregate, got {child:?}"); + }; + assert_eq!(inner_reduction, &Reduction::PerEntity); + assert!(matches!(inner_measures.as_slice(), [AggIntent::Max { .. }])); + assert!(matches!( + inner_child.expect_non_asap(), + NonASAPOp::PromqlSubquery { .. } + )); +} + +#[test] +fn nested_subquery_from_prometheus_docs() { + // SEMANTICS (PromQL): the *nested sub-query* example from the official docs + // (): + // + // max_over_time(deriv(rate(distance_covered_total[5s])[30s:5s])[10m:]) + // + // Two stacked sub-queries, each feeding a range-vector function; the outer + // `[10m:]` uses the **default resolution** (no explicit step). Each level + // lowers to its own node, so the whole spine pins as: + // + // Max ∘ PromqlSubquery{10m, res: None} ∘ Deriv ∘ PromqlSubquery{30s, res: 5s} + // ∘ Rate ∘ TimeRange{5s} ∘ Scan + // + // Every reduction is per-series (no grouping), so the output schema stays + // the label-preserving `[ts, value]`. + let qe = ok("max_over_time(deriv(rate(distance_covered_total[5s])[30s:5s])[10m:])"); + + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected `max_over_time` Aggregate at the root, got {qe:?}"); + }; + assert_eq!(reduction, &Reduction::PerEntity); + assert!(matches!(measures.as_slice(), [AggIntent::Max { .. }])); + + let NonASAPOp::PromqlSubquery { + range, + resolution, + child, + } = child.expect_non_asap() + else { + panic!("expected the outer `[10m:]` PromqlSubquery, got {child:?}"); + }; + assert_eq!(*range, Duration::from_secs(600)); + assert_eq!(*resolution, None, "`[10m:]` keeps the default resolution"); + + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = child.expect_non_asap() + else { + panic!("expected the `deriv` Aggregate, got {child:?}"); + }; + assert_eq!(reduction, &Reduction::PerEntity); + assert!(matches!(measures.as_slice(), [AggIntent::Deriv])); + + let NonASAPOp::PromqlSubquery { + range, + resolution, + child, + } = child.expect_non_asap() + else { + panic!("expected the inner `[30s:5s]` PromqlSubquery, got {child:?}"); + }; + assert_eq!(*range, Duration::from_secs(30)); + assert_eq!(*resolution, Some(Duration::from_secs(5))); + + let NonASAPOp::Aggregate { + measures, child, .. + } = child.expect_non_asap() + else { + panic!("expected the `rate` Aggregate, got {child:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Rate])); + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { + panic!("expected the `[5s]` TimeRange under rate, got {child:?}"); + }; + assert_eq!(*range, Duration::from_secs(5)); + + // Per-series end to end: the schema keeps the (ts, value) floor and stays open. + let schema = qe.schema.clone(); + assert_eq!( + schema + .fields + .iter() + .map(|c| c.name.as_str()) + .collect::>(), + vec!["ts", "value"], + ); + assert!(!schema.closed, "per-series chain never freezes the schema"); +} + +// ───────────────────────────────────────────────────────────────────────────── +// K. Time-shift modifiers (basics §Offset/@; at_modifier.test) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn offset_modifier_lowers_to_a_time_shift() { + // SEMANTICS (PromQL, issue #40): `offset 5m` shifts the lookback 5m into the + // past — a `TimeShift` wrapper over the selector (signed ms; a negative + // offset shifts forward). Schema is unchanged (the shift only moves *when*). + let qe = ok("http_requests_total offset 5m"); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { + panic!("expected an ingestion TimeRange, got {qe:?}"); + }; + let NonASAPOp::TimeShift { shift, child } = child.expect_non_asap() else { + panic!("expected a TimeShift, got {qe:?}"); + }; + assert_eq!(shift.offset_ms, 300_000); + assert!(shift.at.is_none()); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); + + // `offset -5m` shifts forward → negative ms. + let qe = ok("http_requests_total offset -5m"); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { + panic!("expected an ingestion TimeRange"); + }; + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { + panic!("expected a TimeShift"); + }; + assert_eq!(shift.offset_ms, -300_000); +} + +#[test] +fn at_modifier_lowers_to_a_time_shift() { + // SEMANTICS (PromQL, issue #40): `@ ` pins the evaluation to an absolute + // instant (PromQL seconds → IR milliseconds); `@ start()` / `@ end()` anchor + // to the query range bounds. + let qe = ok("http_requests_total @ 1609746000"); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { + panic!("expected an ingestion TimeRange"); + }; + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { + panic!("expected a TimeShift for `@ `"); + }; + assert_eq!(shift.at, Some(AtModifier::Timestamp(1_609_746_000_000))); + assert_eq!(shift.offset_ms, 0); + + let qe = ok("http_requests_total @ start()"); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { + panic!("expected an ingestion TimeRange"); + }; + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { + panic!("expected a TimeShift for `@ start()`"); + }; + assert_eq!(shift.at, Some(AtModifier::Start)); + + // Offset and `@` compose: `@ end() offset 5m` carries both. + let qe = ok("http_requests_total @ end() offset 5m"); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { + panic!("expected an ingestion TimeRange, got {qe:?}"); + }; + let NonASAPOp::TimeShift { shift, .. } = child.expect_non_asap() else { + panic!("expected a TimeShift, got {qe:?}"); + }; + assert_eq!(shift.at, Some(AtModifier::End)); + assert_eq!(shift.offset_ms, 300_000); +} + +#[test] +fn offset_on_a_ranged_selector_wraps_inside_the_time_range() { + // `rate(m[5m] offset 1h)` — the offset is on the ranged selector, so the + // `TimeShift` sits *under* the `TimeRange` (the 5m window is taken at the + // shifted time), and the whole thing under the per-series `Rate` (#40). + let qe = ok("rate(http_requests_total[5m] offset 1h)"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected the rate Aggregate, got {qe:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Rate])); + let NonASAPOp::TimeRange { child, .. } = child.expect_non_asap() else { + panic!("expected a TimeRange under rate, got {child:?}"); + }; + let NonASAPOp::TimeShift { shift, child } = child.expect_non_asap() else { + panic!("expected a TimeShift under the TimeRange, got {child:?}"); + }; + assert_eq!(shift.offset_ms, 3_600_000); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); +} + +// ───────────────────────────────────────────────────────────────────────────── +// L. Unsupported functions (functions.test) — clean rejection +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn unsupported_functions_are_rejected() { + // These parse fine but have no intent-algebra lowering yet. Each must return + // a clean LoweringError rather than mislower. + for q in [ + "step()", + "range()", + r#"histogram_quantiles("le", 0.5, 0.9, x)"#, + // NOTE: counter-derivatives (#44), math/trig (#45, §O), presence (#47, + // §P), time/calendar (#46, §Q), vector/scalar (#48, §R), + // label_replace/label_join (#50, §T) and the extra range reducers + + // sort family (#51, §U) now lower — see those sections. `info` (#84), + // `min_of`/`max_of` (#89) are pinned in §R / §U. + ] { + let _ = rejected(q); + } +} + +// ───────────────────────────────────────────────────────────────────────────── +// M. Counter-derivative range functions (functions.test; issue #44) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn count_over_time_value_column_is_float64() { + // #69: a per-series range reduction produces a PromQL sample value, which is + // always float64. `count_over_time`'s `Count` intent types `Int64`, but the + // derived `value` column must be `Float64` like every other range reducer. + let schema = ok("count_over_time(m[5m])").schema.clone(); + let value = schema + .fields + .iter() + .find(|c| c.name == "value") + .expect("value column"); + assert_eq!(value.dtype, DataType::Float64); +} + +#[test] +fn counter_derivative_functions_lower_to_distinct_intents() { + // Each range function reduces one series' window to one value per series + // (label-preserving), riding on a `TimeRange`, and carries its OWN intent — + // deliberately not aliased to rate/increase/count. + for (q, want) in [ + ("changes(m[15m])", AggIntent::Changes), + ("delta(m[5m])", AggIntent::Delta), + ("idelta(m[5m])", AggIntent::IDelta), + ("deriv(m[1h])", AggIntent::Deriv), + ("resets(m[1h])", AggIntent::Resets), + ] { + let qe = ok(q); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected an Aggregate for {q:?}, got {qe:?}"); + }; + assert_eq!( + reduction, + &Reduction::PerEntity, + "{q}: per-series, no grouping" + ); + assert_eq!( + measures.as_slice(), + std::slice::from_ref(&want), + "{q}: wrong intent" + ); + assert!( + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { .. }), + "{q}: reduction rides on a TimeRange, got {child:?}" + ); + } +} + +#[test] +fn predict_linear_carries_horizon_seconds() { + // `predict_linear(v[w], t)` — the 2nd (scalar) arg is the prediction horizon + // in seconds; it must be carried in the intent (it changes the result). + let qe = ok("predict_linear(node_filesystem_avail_bytes[3h], 86400)"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected an Aggregate, got {qe:?}"); + }; + assert_eq!( + measures.as_slice(), + &[AggIntent::PredictLinear { seconds: 86400.0 }] + ); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); +} + +#[test] +fn double_exponential_smoothing_carries_factors() { + let want = AggIntent::DoubleExpSmoothing { + smoothing: 0.5, + trend: 0.3, + }; + let a = ok("double_exponential_smoothing(m[10m], 0.5, 0.3)"); + assert_eq!(intents(&a).as_slice(), std::slice::from_ref(&want)); +} + +#[test] +fn aggregation_over_counter_derivative_keeps_labels() { + // A counter-derivative is per-series (label-preserving), so an outer + // `sum by (job)` can group on a label the inner `changes` preserved. + let qe = ok(r#"sum by (job) (changes(m{job="api"}[15m]))"#); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate, got {qe:?}"); + }; + assert!( + matches!(reduction, Reduction::Reduce(by) if !by.is_empty()), + "outer sum groups on job" + ); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + assert!(intents(&qe).iter().any(|i| matches!(i, AggIntent::Changes))); + let _ = child; +} + +#[test] +fn outer_stat_over_counter_derivative_nests_two_levels() { + // A cross-series stat over a counter-derivative is a genuine two-level + // reduction: the derivative runs per series (inner), the stat aggregates + // across series (outer). They must not collapse into one node — and a + // grouped outer (`avg by (dc)`) must resolve its key against the labels the + // inner reduction preserved, threading any scalar param (predict horizon). + let qe = ok("avg by (dc) (predict_linear(m[3h], 3600))"); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate, got {qe:?}"); + }; + assert!( + matches!(reduction, Reduction::Reduce(by) if !by.is_empty()), + "outer `avg by (dc)` groups on a label" + ); + assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); + let NonASAPOp::Aggregate { + reduction: inner_reduction, + measures: inner_measures, + .. + } = child.expect_non_asap() + else { + panic!("expected inner per-series Aggregate, got {child:?}"); + }; + assert_eq!( + inner_reduction, + &Reduction::PerEntity, + "inner derivative stays per-series" + ); + assert_eq!( + inner_measures.as_slice(), + std::slice::from_ref(&AggIntent::PredictLinear { seconds: 3600.0 }) + ); +} + +#[test] +fn topk_over_counter_derivative_is_generic_sort_limit() { + // `topk(k, deriv(...))` ranks the per-series derivative values — a generic + // `Sort + Limit`, NOT a heavy-hitter `TopK` (that's only `count_over_time`). + let qe = ok("topk(3, deriv(m[5m]))"); + let NonASAPOp::Limit { + n: Some(n), child, .. + } = qe.expect_non_asap() + else { + panic!("expected Limit, got {qe:?}"); + }; + assert_eq!(*n, 3); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Sort { .. })); + assert!(intents(&qe).iter().any(|i| matches!(i, AggIntent::Deriv))); + assert!( + !intents(&qe) + .iter() + .any(|i| matches!(i, AggIntent::TopK { .. })), + "counter-derivative topk is generic ranking, not a heavy-hitter sketch" + ); +} + +#[test] +fn counter_derivative_composes_in_binary_ops() { + // As a vector operand: `delta(a[5m]) / delta(b[5m])` is a BinaryOp of two + // per-series Delta reductions. + let ratio = ok("delta(a[5m]) / delta(b[5m])"); + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + lhs, + rhs, + .. + } = ratio.expect_non_asap() + else { + panic!("expected BinaryOp, got {ratio:?}"); + }; + assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)); + assert!( + matches!(lhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) + ); + assert!( + matches!(rhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if measures.as_slice() == [AggIntent::Delta]) + ); + + // Under an aggregate over a binary op mixing a counter-derivative with + // another per-series function: `sum(rate(m[5m]) + changes(m[5m]))`. + let mixed = ok("sum(rate(m[5m]) + changes(m[5m]))"); + let NonASAPOp::Aggregate { + measures, child, .. + } = mixed.expect_non_asap() + else { + panic!("expected Aggregate, got {mixed:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::BinaryOp { .. } + )); + assert!(intents(&mixed).iter().any(|i| matches!(i, AggIntent::Rate))); + assert!(intents(&mixed) + .iter() + .any(|i| matches!(i, AggIntent::Changes))); +} + +#[test] +fn range_functions_over_a_subquery_reduce_per_series() { + // Issue #55 — the whole range-vector family accepts a sub-query argument + // (generalizing `*_over_time`, #42): `rate`/`increase`/`irate` and the + // counter-derivatives. Each lowers to a per-series `Aggregate{[f]}` directly + // over the `PromqlSubquery` — the sub-query is the range context, so there is NO + // separate `TimeRange` (that would double the range). + for (q, want) in [ + ("rate(sum(m)[5m:])", AggIntent::Rate), + ("increase(sum(m)[5m:])", AggIntent::Increase), + ("irate(sum(m)[5m:])", AggIntent::IRate), + ("changes(rate(m[5m])[1h:])", AggIntent::Changes), + ("delta(sum(m)[5m:])", AggIntent::Delta), + ("deriv(sum(m)[10m:])", AggIntent::Deriv), + ("resets(sum(m)[5m:])", AggIntent::Resets), + ] { + let qe = ok(q); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("{q}: expected an Aggregate, got {qe:?}"); + }; + assert_eq!( + reduction, + &Reduction::PerEntity, + "{q}: per-series, no grouping" + ); + assert_eq!( + measures.as_slice(), + std::slice::from_ref(&want), + "{q}: wrong intent" + ); + assert!( + matches!(child.expect_non_asap(), NonASAPOp::PromqlSubquery { .. }), + "{q}: reduces directly over the PromqlSubquery (no TimeRange), got {child:?}" + ); + } +} + +#[test] +fn predict_linear_and_double_exp_over_a_subquery_carry_params() { + // The scalar params survive the sub-query path. + let pl = ok("predict_linear(sum(m)[1h:], 3600)"); + assert!(intents(&pl).iter().any( + |i| matches!(i, AggIntent::PredictLinear { seconds } if (*seconds - 3600.0).abs() < 1e-9) + )); + let de = ok("double_exponential_smoothing(sum(m)[10m:], 0.5, 0.3)"); + assert!(intents(&de).iter().any(|i| matches!( + i, + AggIntent::DoubleExpSmoothing { smoothing, trend } + if (*smoothing - 0.5).abs() < 1e-9 && (*trend - 0.3).abs() < 1e-9 + ))); +} + +// ───────────────────────────────────────────────────────────────────────────── +// N. Native-histogram accessors (functions.test; issue #43) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn histogram_quantile_classic_bucket_vs_native() { + // Two lowerings of `histogram_quantile(φ, …)`: the classic cumulative-bucket + // form → exact `HistogramQuantile`; native samples require a new type. + // The classic form is recognised by + // `by (le)`, a `_bucket` metric, or an `le` matcher (issue #43). + for classic in [ + "histogram_quantile(0.9, sum by (le) (rate(x_bucket[5m])))", + "histogram_quantile(0.9, rate(x_bucket[5m]))", // bare _bucket metric + r#"histogram_quantile(0.9, rate(x{le="0.5"}[5m]))"#, // le matcher + ] { + let qe = ok(classic); + assert!( + has( + &qe, + |i| matches!(i, AggIntent::HistogramQuantile { q, .. } if (*q - 0.9).abs() < 1e-9) + ), + "classic bucket form → HistogramQuantile: {classic}" + ); + assert!( + !has(&qe, |i| matches!(i, AggIntent::Quantile { .. })), + "{classic}" + ); + } + for native in [ + "histogram_quantile(0.9, my_native_histogram)", + "histogram_quantile(0.9, request_duration_seconds)", // raw samples (your extension) + ] { + rejected(native); + } +} + +#[test] +fn native_histogram_accessors_are_explicit_gaps() { + // Native histogram samples have no typed representation yet. + for q in [ + "histogram_count(v)", + "histogram_sum(v)", + "histogram_avg(v)", + "histogram_stddev(v)", + "histogram_stdvar(v)", + ] { + rejected(q); + } +} + +#[test] +fn histogram_fraction_is_an_explicit_gap() { + rejected("histogram_fraction(0, 0.2, v)"); +} + +// ───────────────────────────────────────────────────────────────────────────── +// O. Math / trig scalar-transform functions (functions.test; issue #45) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn math_functions_lower_to_typed_scalar_projections() { + for name in [ + "abs", "ceil", "floor", "sqrt", "ln", "log2", "sgn", "sin", "atanh", "deg", "rad", + ] { + let query = ok(&format!("{name}(v)")); + assert!( + matches!(support::sample_expression(&query),ScalarExpr::FunctionCall { name:n,args } if n==&format!("promql_{name}") && args.len()==1) + ); + query.validate_structure().unwrap(); + } +} + +#[test] +fn clamp_and_round_carry_their_params() { + for (query, params) in [ + ("clamp(v,0,100)", vec![0.0, 100.0]), + ("clamp_min(v,1)", vec![1.0]), + ("clamp_max(v,5)", vec![5.0]), + ("round(v)", vec![1.0]), + ("round(v,5)", vec![5.0]), + ] { + let node = ok(query); + let ScalarExpr::FunctionCall { args, .. } = support::sample_expression(&node) else { + panic!() + }; + assert_eq!( + args.iter().skip(1).map(promql_scalar).collect::>(), + params.into_iter().map(Some).collect::>() + ); + } +} + +#[test] +fn pi_lowers_to_a_scalar_constant() { + // `pi()` is the constant π — a `ScalarExpr` leaf, not a `Math` intent. + assert!(promql_scalar(&support::scalar_root("pi()")) + .is_some_and(|v| (v - std::f64::consts::PI).abs() < 1e-12)); +} + +// ───────────────────────────────────────────────────────────────────────────── +// P. Presence functions (functions.test; issue #47) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn presence_functions_lower_to_presence_intents() { + for (q, want) in [ + (r#"absent(up{job="x"})"#, AggIntent::Absent), + ("absent_over_time(m[1h])", AggIntent::AbsentOverTime), + ("present_over_time(m[5m])", AggIntent::PresentOverTime), + ] { + let qe = ok(q); + assert!(intents(&qe).contains(&want), "{q}: got {:?}", intents(&qe)); + } +} + +#[test] +fn absent_keeps_matcher_labels_for_the_synthesized_output() { + // `absent(v)` synthesizes its output labels from `v`'s equality matchers, so + // those labels must survive into the schema — here `job` from `{job="x"}`. + let qe = ok(r#"absent(up{job="x"})"#); + let cols = qe.schema.clone(); + assert!( + cols.fields.iter().any(|c| c.name == "job"), + "matcher label `job` kept, got {:?}", + cols.fields.iter().map(|c| &c.name).collect::>() + ); +} + +// ───────────────────────────────────────────────────────────────────────────── +// Q. Time / calendar functions (functions.test; issue #46) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn time_lowers_to_the_eval_time_scalar() { + assert!(matches!( + support::scalar_root("time()"), + ScalarExpr::EvalTimestamp + )); +} + +#[test] +fn time_minus_vector_is_the_uptime_pattern() { + let qe = ok("time() - process_start_time_seconds"); + assert!( + matches!(support::sample_expression(&qe), ScalarExpr::Arithmetic { op: ArithmeticOpKind::Sub, left, .. } if matches!(left.as_ref(), ScalarExpr::EvalTimestamp)) + ); + assert!(qe.schema.time_index.is_some()); +} + +#[test] +fn calendar_functions_lower_to_time_fn_intents() { + assert!(has(&ok("timestamp(up)"), |i| *i + == AggIntent::TimeFn(TimeFunc::Timestamp))); + for name in [ + "minute", + "hour", + "day_of_week", + "day_of_month", + "day_of_year", + "month", + "year", + "days_in_month", + ] { + let query = ok(&format!("{name}(v)")); + assert!( + matches!(support::sample_expression(&query),ScalarExpr::FunctionCall { name:n,args } if n==&format!("promql_{name}") && args.len()==1) + ); + } +} + +#[test] +fn no_arg_calendar_function_reads_the_eval_time() { + let query = ok("day_of_week()"); + let NonASAPOp::Project { child, .. } = query.expect_non_asap() else { + panic!() + }; + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::PromqlVectorFromScalar(ScalarExpr::EvalTimestamp) + )); + assert!( + matches!(support::sample_expression(&query),ScalarExpr::FunctionCall { name,.. } if name=="promql_day_of_week") + ); +} + +#[test] +fn timestamp_composes_under_an_outer_aggregation() { + // `sum by (job) (timestamp(up))` — the per-series `timestamp` transform sits + // below an ordinary grouped sum. Both intents must appear in the tree. + let qe = ok("sum by (job) (timestamp(up))"); + assert!(has(&qe, |i| *i == AggIntent::TimeFn(TimeFunc::Timestamp))); + assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); +} + +// ───────────────────────────────────────────────────────────────────────────── +// R. Type-conversion functions: vector() / scalar() (functions.test; issue #48) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn vector_promotes_a_scalar_to_a_vector() { + // SEMANTICS: `vector(s)` is the scalar→instant-vector bridge — a label-less + // single series carrying the scalar's value. + let qe = ok("vector(1)"); + let NonASAPOp::PromqlVectorFromScalar(inner) = qe.expect_non_asap() else { + panic!("expected PromqlVectorFromScalar, got {qe:?}"); + }; + assert!(matches!(inner, ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); + // Vector-typed: schema has a time index (a scalar leaf has none). + let sch = qe.schema.clone(); + assert!(sch.time_index.is_some()); + assert!(sch.fields.iter().any(|c| c.name == "value")); +} + +#[test] +fn scalar_collapses_a_vector_to_a_scalar() { + let qe = support::scalar_root("scalar(node_load1)"); + let ScalarExpr::PromqlScalarFromVector(inner) = &qe else { + panic!() + }; + assert_eq!(first_scan(inner).0, "node_load1"); +} + +#[test] +fn vector_zero_is_a_vector_operand_of_a_set_op() { + // `up or vector(0)` — the dead-man's-switch. `or` is a set op between two + // vectors, so `vector(0)` must be a vector (a `PromqlVectorFromScalar`), never a + // folded scalar operand. + let qe = ok("up or vector(0)"); + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + rhs, + .. + } = qe.expect_non_asap() + else { + panic!("expected a BinaryOp, got {qe:?}"); + }; + assert_eq!(*op, BinaryOpKind::Set(PromQLVectorSetOpKind::Or)); + assert!(matches!( + rhs.expect_non_asap(), + NonASAPOp::PromqlVectorFromScalar(_) + )); +} + +#[test] +fn scalar_of_a_vector_feeds_a_threshold_comparison() { + let qe = ok("node_load1 > scalar(node_cpu_count)"); + let ScalarExpr::Compare { right, .. } = support::sample_expression(&qe) else { + panic!() + }; + assert!(matches!( + right.as_ref(), + ScalarExpr::PromqlScalarFromVector(_) + )); + assert!(qe.schema.time_index.is_some()); +} + +#[test] +fn info_lowers_to_a_label_enrichment_join() { + // `info(v, [selector])` is a label-enrichment *join* against the info + // metric(s) — it lowers to an `PromqlInfoEnrich` over the (unchanged) input vector + // (issue #84). The value/time axis pass through; the enriched labels are + // runtime, so the schema stays the child's. + let qe = ok("info(rate(http_requests_total[5m]))"); + let NonASAPOp::PromqlInfoEnrich { selector, child } = qe.expect_non_asap() else { + panic!("expected an PromqlInfoEnrich, got {qe:?}"); + }; + assert!(selector.is_empty(), "no selector → default target_info"); + // The child is the untouched input (a per-series rate reduction here). + assert!(has(child, |i| *i == AggIntent::Rate)); + assert!(qe.schema.clone().time_index.is_some()); +} + +#[test] +fn info_selector_carries_the_info_side_matchers() { + // `info(v, {__name__=~".+_info", data=~".+"})` — the selector picks the info + // metric(s) via `__name__` and constrains the data labels. Regex / `__name__` + // matchers are kept symbolically (not run through the single-metric selector + // path). + let qe = ok(r#"info(build_info, {__name__=~".+_info", another_data=~".+"})"#); + let NonASAPOp::PromqlInfoEnrich { selector, .. } = qe.expect_non_asap() else { + panic!("expected an PromqlInfoEnrich, got {qe:?}"); + }; + assert_eq!( + selector.len(), + 2, + "both selector matchers kept: {selector:?}" + ); + assert!(selector + .iter() + .any(|m| m.label == "__name__" && m.op == CompareOpKind::Regex)); + assert!(selector.iter().any(|m| m.label == "another_data")); +} + +#[test] +fn info_composes_under_an_aggregation_and_over_a_time_shift() { + // `sum(info(m))` — enrichment first, then a cross-series sum over it. + assert!(has(&ok("sum(info(node_uname_info))"), |i| matches!( + i, + AggIntent::Sum { .. } + ))); + // `offset` / `@` on the input now lower to a `TimeShift` under the info-join + // (issue #40) — the enrichment composes over the shifted selector. + assert!(matches!( + ok("info(metric @ 60)").expect_non_asap(), + NonASAPOp::PromqlInfoEnrich { .. } + )); + assert!(matches!( + ok("info(metric offset 1m)").expect_non_asap(), + NonASAPOp::PromqlInfoEnrich { .. } + )); +} + +// ───────────────────────────────────────────────────────────────────────────── +// S. Extended aggregation operators: group / count_values (aggregators.test; #49) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn group_lowers_to_a_constant_group_intent() { + // SEMANTICS: `group(v)` yields a constant 1 per group — a distinct intent, + // NOT folded onto `sum` (which would return the value sum instead of 1). + let qe = ok("group(up)"); + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { + panic!("expected an Aggregate, got {qe:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Group])); + // Output column is the constant-1 `group` value. + let sch = qe.schema.clone(); + assert!(sch.fields.iter().any(|c| c.name == "group")); +} + +#[test] +fn group_by_keeps_the_grouping_keys() { + // `group by (job) (up)` — the grouping keys ride on `Aggregate.by`. + let qe = ok("group by (job) (up)"); + let sch = qe.schema.clone(); + assert!(sch.fields.iter().any(|c| c.name == "job")); + assert!(has(&qe, |i| *i == AggIntent::Group)); +} + +#[test] +fn count_values_groups_by_value_and_synthesizes_a_label() { + // SEMANTICS: `count_values("l", v)` groups the input series by their sample + // value, counts each distinct value, and emits that value as a new label + // `l`. The intent carries the label; schema gains a `Utf8` `l` column. + let qe = ok(r#"count_values("version", build_version)"#); + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { + panic!("expected an Aggregate, got {qe:?}"); + }; + assert!( + matches!(measures.as_slice(), [AggIntent::CountValues { label }] if label == "version") + ); + let sch = qe.schema.clone(); + let version = sch + .fields + .iter() + .find(|c| c.name == "version") + .expect("synthesized `version` label column"); + assert_eq!( + version.dtype, + DataType::Utf8, + "the value becomes a string label" + ); + assert!( + sch.fields.iter().any(|c| c.name == "count"), + "and a count column" + ); +} + +#[test] +fn count_values_accepts_a_parenthesised_label_and_by_grouping() { + // `count_values by (job) ((("v")), m)` — nested parens around the string + // param, plus `by` grouping. Both survive. + let qe = ok(r#"count_values by (job) ((("v")), m)"#); + assert!(has( + &qe, + |i| matches!(i, AggIntent::CountValues { label } if label == "v") + )); + let sch = qe.schema.clone(); + assert!(sch.fields.iter().any(|c| c.name == "job")); + assert!(sch.fields.iter().any(|c| c.name == "v")); +} + +#[test] +fn count_values_label_colliding_with_a_group_key_is_not_duplicated() { + // `count_values by (job)("job", v)` — the synthesized label name collides + // with a group-by key. PromQL's synthesized label takes precedence; the + // output must carry a single `job` column, never two. + let qe = ok(r#"count_values by (job) ("job", version)"#); + let sch = qe.schema.clone(); + let jobs = sch.fields.iter().filter(|c| c.name == "job").count(); + assert_eq!(jobs, 1, "collision deduped, got {:?}", sch.fields); + assert!(sch.fields.iter().any(|c| c.name == "count")); +} + +#[test] +fn limitk_and_limit_ratio_lower_to_series_sampling() { + // `limitk`/`limit_ratio` are series-*sampling* selection — a subset of whole + // series kept unchanged (NOT a ranking), so they lower to the dedicated + // `PromqlSeriesSample` node, never `topk`'s `Sort → Limit` (issue #86). + assert!(matches!( + ok("limitk(2, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { + kind: SampleKind::LimitK(2), + .. + } + )); + assert!(matches!( + ok("limit_ratio(0.1, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 0.1).abs() < 1e-9 + )); + // Series-preserving: the output schema equals the input's (ts, value). + let sch = ok("limitk(2, http_requests)").schema.clone(); + assert!(sch.fields.iter().any(|c| c.name == "value")); + assert!(sch.time_index.is_some()); +} + +#[test] +fn limit_ratio_keeps_a_negative_ratio_and_clamps_out_of_range() { + // A negative ratio selects the complementary fraction — it must survive, not + // be normalised away. Out-of-range magnitudes clamp to [-1, 1] (Prometheus). + assert!(matches!( + ok("limit_ratio(-0.5, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r + 0.5).abs() < 1e-9 + )); + assert!(matches!( + ok("limit_ratio(1.1, http_requests)").expect_non_asap(), + NonASAPOp::PromqlSeriesSample { kind: SampleKind::LimitRatio(r), .. } if (r - 1.0).abs() < 1e-9 + )); +} + +#[test] +fn limitk_by_carries_the_grouping_and_composes_in_a_set_op() { + // `limitk by (group)` samples per group; the grouping label is seeded. + let qe = ok("limitk by (group) (2, http_requests)"); + let NonASAPOp::PromqlSeriesSample { by, .. } = qe.expect_non_asap() else { + panic!("expected a PromqlSeriesSample, got {qe:?}"); + }; + assert!(!by.is_empty(), "grouped sampling keeps its `by` keys"); + // `count(limitk(2, v) and v)` — the surviving series' identity matters, so + // the PromqlSeriesSample must be preserved under the set op (it must lower, not reject). + assert!(has( + &ok("count(limitk(2, http_requests) and http_requests)"), + |i| matches!(i, AggIntent::Count { .. }) + )); +} + +#[test] +fn dynamic_and_non_finite_sample_params_are_rejected() { + // A dynamic k/ratio (not a compile-time constant) or a NaN can't be a static + // `PromqlSeriesSample` param — rejected rather than mislowered. + let _ = rejected("limitk(NaN, http_requests)"); + let _ = rejected("limitk(scalar(foo), http_requests)"); + let _ = rejected("limit_ratio(time() % 17 / 17, http_requests)"); +} + +// ───────────────────────────────────────────────────────────────────────────── +// T. Label-rewrite functions: label_replace / label_join (functions.test; #50) +// ───────────────────────────────────────────────────────────────────────────── + +/// Descend single-child nodes to the first `PromqlRelabel`. +fn first_relabel(e: &OperatorNode) -> &OperatorNode { + match e.expect_non_asap() { + NonASAPOp::PromqlRelabel { .. } => e, + NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::TimeRange { child, .. } + | NonASAPOp::TimeShift { child, .. } => first_relabel(child), + other => panic!("no PromqlRelabel reachable from {other:?}"), + } +} + +/// True when `value` is a `FunctionCall` with the given name. +fn is_fn_named(value: &ScalarExpr, name: &str) -> bool { + matches!(value, ScalarExpr::FunctionCall { name: n, .. } if n == name) +} + +#[test] +fn label_replace_is_a_relabel_over_the_vector() { + // SEMANTICS: `label_replace(v, dst, repl, src, regex)` rewrites the `dst` + // label per series from a regex over `src`; the sample value is untouched. + let qe = ok(r#"label_replace(up, "host", "$1", "instance", "(.+):.*")"#); + let NonASAPOp::PromqlRelabel { dst, value, child } = qe.expect_non_asap() else { + panic!("expected a PromqlRelabel, got {qe:?}"); + }; + assert_eq!(dst, "host"); + // The child is the untouched vector. + let (metric, _) = first_scan(child); + assert_eq!(metric, "up"); + // The value expression is a `label_replace` fn reading the `src` label. + assert!(is_fn_named(value, "label_replace")); + // Output: the child's columns + the synthesized `host` label; value & ts kept. + let sch = qe.schema.clone(); + assert!(sch.fields.iter().any(|c| c.name == "host")); + assert!(sch.fields.iter().any(|c| c.name == "value")); + assert!(sch.time_index.is_some(), "the vector's time axis survives"); +} + +#[test] +fn label_join_concatenates_source_labels() { + // SEMANTICS: `label_join(v, dst, sep, src…)` joins the source labels with + // `sep` into `dst`. + let qe = ok(r#"label_join(up, "combined", "-", "job", "instance")"#); + let NonASAPOp::PromqlRelabel { dst, value, .. } = qe.expect_non_asap() else { + panic!("expected a PromqlRelabel, got {qe:?}"); + }; + assert_eq!(dst, "combined"); + assert!(is_fn_named(value, "label_join")); + let sch = qe.schema.clone(); + assert!(sch.fields.iter().any(|c| c.name == "combined")); +} + +#[test] +fn label_replace_composes_under_an_aggregation() { + // `sum by (host) (label_replace(up, "host", "$1", "instance", "(.+):.*"))` — + // relabel first, then group by the synthesized label. + let qe = ok(r#"sum by (host) (label_replace(up, "host", "$1", "instance", "(.+):.*"))"#); + // A PromqlRelabel sits below the outer Sum. + let relabel = first_relabel(&qe); + assert!( + matches!(relabel.expect_non_asap(), NonASAPOp::PromqlRelabel { dst, .. } if dst == "host") + ); + assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); + let sch = qe.schema.clone(); + assert!(sch.fields.iter().any(|c| c.name == "host")); +} + +// ───────────────────────────────────────────────────────────────────────────── +// U. Long-tail: extra range reducers + the sort family (functions.test; #51) +// ───────────────────────────────────────────────────────────────────────────── + +#[test] +fn extra_over_time_reducers_lower_to_per_series_intents() { + // SEMANTICS: each is a per-series reduction of one series' range window to a + // single value — a `TimeRange`-wrapped `Aggregate` with the matching intent. + for (q, want) in [ + ("last_over_time(m[5m])", AggIntent::LastOverTime), + ("first_over_time(m[5m])", AggIntent::FirstOverTime), + ("mad_over_time(m[5m])", AggIntent::MadOverTime), + ("ts_of_min_over_time(m[5m])", AggIntent::TsOfMinOverTime), + ("ts_of_max_over_time(m[5m])", AggIntent::TsOfMaxOverTime), + ("ts_of_first_over_time(m[5m])", AggIntent::TsOfFirstOverTime), + ("ts_of_last_over_time(m[5m])", AggIntent::TsOfLastOverTime), + ] { + let qe = ok(q); + assert!(has(&qe, |i| *i == want), "{q}: {:?}", intents(&qe)); + // Per-series: the range window survives as a `TimeRange`. + assert!( + matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::TimeRange { .. })), + "{q} keeps its range as a TimeRange" + ); + } +} + +#[test] +fn last_over_time_composes_under_an_outer_aggregation() { + // `sum by (job) (last_over_time(m[5m]))` — per-series last, THEN cross-series + // sum. Both intents survive (issue #27's arbitrary nesting). + let qe = ok("sum by (job) (last_over_time(m[5m]))"); + assert!(has(&qe, |i| *i == AggIntent::LastOverTime)); + assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); +} + +#[test] +fn sort_and_sort_desc_reorder_by_value_without_a_limit() { + // SEMANTICS: `sort`/`sort_desc` reorder an instant vector by sample value. + // Row-preserving → a bare `Sort` (no `Limit`), ascending / descending. + for (q, ascending) in [ + ("sort(http_requests)", true), + ("sort_desc(http_requests)", false), + ] { + let qe = ok(q); + let NonASAPOp::Sort { keys, child, .. } = qe.expect_non_asap() else { + panic!("{q}: expected a Sort, got {qe:?}"); + }; + assert_eq!(keys.len(), 1); + assert_eq!(keys[0].ascending, ascending, "{q}"); + // No Limit above the Sort — every series is preserved. + assert!(!matches!(qe.expect_non_asap(), NonASAPOp::Limit { .. })); + // The value column is what it ranks on: descend to the scan. + let (metric, _) = first_scan(child); + assert_eq!(metric, "http_requests"); + } +} + +#[test] +fn sort_by_label_orders_on_each_label_in_turn() { + // `sort_by_label(v, "group", "instance", "job")` — one ascending sort key per + // label, in argument order; the labels are seeded into the schema. + let qe = ok(r#"sort_by_label(http_requests, "group", "instance", "job")"#); + let NonASAPOp::Sort { keys, .. } = qe.expect_non_asap() else { + panic!("expected a Sort, got {qe:?}"); + }; + assert_eq!(keys.len(), 3, "one key per label"); + assert!(keys.iter().all(|k| k.ascending)); + let sch = qe.schema.clone(); + for label in ["group", "instance", "job"] { + assert!(sch.fields.iter().any(|c| c.name == label), "{label} seeded"); + } +} + +#[test] +fn sort_by_label_desc_is_descending() { + let qe = ok(r#"sort_by_label_desc(http_requests, "instance")"#); + let NonASAPOp::Sort { keys, .. } = qe.expect_non_asap() else { + panic!("expected a Sort, got {qe:?}"); + }; + assert!(keys.iter().all(|k| !k.ascending)); +} + +#[test] +fn min_of_max_of_fold_constant_scalars() { + // `min_of`/`max_of` are n-ary scalar reducers. When every argument is a + // constant they constant-fold to a `ScalarExpr` leaf, just like scalar + // arithmetic (#35) — the only form the intent algebra can hold (#89). + assert_eq!( + promql_scalar(&support::scalar_root("min_of(3, 5)")), + Some(3.0) + ); + assert_eq!( + promql_scalar(&support::scalar_root("max_of(3, 5)")), + Some(5.0) + ); + assert_eq!( + promql_scalar(&support::scalar_root("min_of(-2, -5)")), + Some(-5.0) + ); + // Nested folds and use as a threshold operand. + assert_eq!( + promql_scalar(&support::scalar_root("max_of(min_of(2, 3), 10)")), + Some(10.0) + ); + let qe = ok("up > max_of(1, 2)"); + let ScalarExpr::Compare { right: rhs, .. } = support::sample_expression(&qe) else { + panic!("{qe:?}") + }; + assert_eq!(promql_scalar(rhs), Some(2.0)); +} + +#[test] +fn min_of_max_of_ignore_nan_like_the_min_max_aggregators() { + // A NaN argument is skipped (Prometheus `min`/`max` NaN semantics). + assert_eq!( + promql_scalar(&support::scalar_root("max_of(3, NaN)")), + Some(3.0) + ); + assert_eq!( + promql_scalar(&support::scalar_root("min_of(NaN, 3)")), + Some(3.0) + ); +} + +#[test] +fn non_constant_min_of_max_of_is_rejected__GAP() { + // A dynamic argument (`step()` — itself unsupported, #89) can't be folded to + // a constant and there is no scalar min/max node, so it stays rejected + // rather than mislowered. These forms also only appear inside unsupported + // dynamic range / offset positions in the corpus. + let _ = rejected("min_of(step(), 1s)"); + let _ = rejected("max_of(min_of(step() + 1, 1h), 1ms)"); +} diff --git a/crates/frontend-promql/tests/unified_promql_lowering.rs b/crates/frontend-promql/tests/unified_promql_lowering.rs new file mode 100644 index 000000000..de769361e --- /dev/null +++ b/crates/frontend-promql/tests/unified_promql_lowering.rs @@ -0,0 +1,1615 @@ +//! End-to-end tests for PromQL → unresolved → canonical tree lowering. + +use std::rc::Rc; +use std::time::Duration; + +use asap_types::ir::{ + BinaryOperator, ExprSemantics, NonASAPOp, OperatorNode, ScalarExpr, TimeRangeKind, +}; +use asap_types::pre_asap::{ + AggIntent, ArithmeticOpKind, BinaryOpKind, CompareOpKind, Reduction, ScalarValue, Source, +}; +use asap_types::types::AccuracyTarget; +use asap_types::workload::{ + AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, + Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, +}; + +use asap_frontend_promql::unified::{lower_promql_workload, PromqlError as LoweringError}; +#[path = "unified_support.rs"] +mod support; +use support::lower_promql; + +fn lower(q: &str) -> Rc { + lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("lower failed for {q:?}: {e}")) +} + +#[test] +fn frequency_extensions_lower_to_explicit_frequency_statistics() { + // ProjectASAP extensions reduce a frequency vector; numeric sample norms + // have different semantics and must never be silently aliased here. + for (query, expected) in [ + ( + "entropy_over_time(cpu_usage[5m])", + AggIntent::FrequencyEntropy { + col: None, + accuracy: AccuracyTarget::Exact, + }, + ), + ( + "l2_over_time(cpu_usage[5m])", + AggIntent::FrequencyL2 { + col: None, + accuracy: AccuracyTarget::Exact, + }, + ), + ] { + assert!(all_intents(&lower(query)) + .iter() + .any(|intent| std::mem::discriminant(intent) == std::mem::discriminant(&expected))); + } +} + +#[test] +fn distinct_over_time_preserves_cardinality_accuracy_and_nested_windows() { + // Distinct counts sample values, not samples or series; all lowering routes + // retain the caller's accuracy requirement, including subquery arguments. + for query in [ + "distinct_over_time(cpu_usage{job=\"worker\"}[5m] offset 1h)", + "distinct_over_time((cpu_usage + 1)[5m:1m])", + "sum by(job)(distinct_over_time(cpu_usage[5m]))", + ] { + for accuracy in [AccuracyTarget::Exact, AccuracyTarget::Epsilon(0.02)] { + let tree = lower_promql(query, accuracy.clone()).unwrap(); + let mut intents = Vec::new(); + collect_intents(&tree, &mut intents); + assert!( + intents.iter().any(|intent| matches!( + intent, AggIntent::Cardinality { accuracy: actual, .. } if actual == &accuracy + )), + "{query}: {tree:?}" + ); + assert!(!intents + .iter() + .any(|intent| matches!(intent, AggIntent::Count { .. }))); + } + } +} + +// ── Bare selectors & label matchers (folded onto Scan.predicates) ─────────────── + +#[test] +fn bare_selector_is_scan_with_predicates() { + let qe = lower(r#"http_requests_total{env="prod",status!="500"}"#); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { + panic!("expected TimeRange, got {qe:?}"); + }; + let NonASAPOp::Scan { + source, predicates, .. + } = child.expect_non_asap() + else { + panic!("expected Scan, got {qe:?}"); + }; + assert!(matches!(source, Source::TimeSeries { metric } if metric == "http_requests_total")); + // The converter splits the matcher conjunction into one predicate per + // conjunct on the Scan. + assert_eq!(predicates.len(), 2); + assert!(predicates + .iter() + .all(|p| matches!(&p.0, ScalarExpr::Compare { .. }))); +} + +#[test] +fn regex_matcher_lowers_to_regex_compareop() { + let qe = lower(r#"http_requests_total{path=~"/api/.*"}"#); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { + panic!("expected TimeRange, got {qe:?}"); + }; + let NonASAPOp::Scan { + predicates, schema, .. + } = child.expect_non_asap() + else { + panic!("expected Scan, got {qe:?}"); + }; + let ScalarExpr::Compare { + left, op, right, .. + } = &predicates[0].0 + else { + panic!("expected Compare, got {:?}", predicates[0].0); + }; + assert_eq!(*op, CompareOpKind::Regex); + // The label matcher's column is resolved positionally against the scan schema. + let path_id = schema.column_id("path").expect("path in scan schema"); + assert!(matches!(left.as_ref(), ScalarExpr::Column(id) if *id == path_id)); + assert!(matches!(right.as_ref(), ScalarExpr::Literal(ScalarValue::Utf8(v)) if v == "/api/.*")); +} + +// ── *_over_time → Aggregate over TimeRange ────────────────────────────────────── + +#[test] +fn quantile_over_time_is_time_range_aggregate() { + let qe = lower(r#"quantile_over_time(0.99, http_request_duration{env="prod"}[5m])"#); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected Aggregate, got {qe:?}"); + }; + assert_eq!(reduction, &Reduction::PerEntity); + assert!( + matches!(measures.as_slice(), [AggIntent::Quantile { q, .. }] if (*q - 0.99).abs() < 1e-9) + ); + let NonASAPOp::TimeRange { range, child, .. } = child.expect_non_asap() else { + panic!("expected TimeRange child, got {child:?}"); + }; + assert_eq!(*range, Duration::from_secs(300)); + // The label matcher folded onto the Scan. + assert!( + matches!(child.expect_non_asap(), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1) + ); +} + +#[test] +fn outer_sum_by_over_quantile_over_time_groups_positionally() { + // `sum by (host) (quantile_over_time(...))`: inner per-series + // quantile-over-time (label-preserving), then an outer cross-series sum + // grouped on a positional `Aggregate.by` — the same shape SQL produces, not + // a name-based Partition. Leaf = [ts, value, host, service] (referenced + // names appended sorted) → host = col 2. + let qe = lower(r#"sum by (host) (quantile_over_time(0.99, latency{service="web"}[5m]))"#); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate grouped by host, got {qe:?}"); + }; + assert_eq!(reduction, &Reduction::by(vec![2])); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + // Inner: Aggregate{Quantile} over TimeRange (per-series over_time reduction). + let NonASAPOp::Aggregate { + measures, child, .. + } = child.expect_non_asap() + else { + panic!("expected Aggregate (quantile_over_time) under the outer Sum, got {child:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Quantile { .. }])); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); +} + +#[test] +fn avg_over_time_maps_to_avg_intent() { + let qe = lower("avg_over_time(cpu_seconds_total[10m])"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected Aggregate, got {qe:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Avg { .. }])); + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { + panic!("expected TimeRange child, got {child:?}"); + }; + assert_eq!(*range, Duration::from_secs(600)); +} + +#[test] +fn stddev_and_stdvar_over_time() { + let qe = lower("stddev_over_time(m[5m])"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected Aggregate"); + }; + assert!(matches!( + measures.as_slice(), + [AggIntent::StdDev { + population: true, + .. + }] + )); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); + + let qe = lower("stdvar_over_time(m[5m])"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected Aggregate"); + }; + assert!(matches!( + measures.as_slice(), + [AggIntent::Variance { + population: true, + .. + }] + )); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); +} + +#[test] +fn histogram_quantile_wraps_inner_in_quantile() { + // The argument's structure (here `rate`) is preserved *under* the quantile, + // not squashed away. The `_bucket` metric + `le` matcher mark the classic + // form → `HistogramQuantile` over `Aggregate{Rate}` over Scan. + let qe = lower(r#"histogram_quantile(0.95, rate(http_duration_seconds_bucket{le="0.5"}[5m]))"#); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate{{HistogramQuantile}}, got {qe:?}"); + }; + assert!( + matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if (*q - 0.95).abs() < 1e-9) + ); + let NonASAPOp::Aggregate { + measures, child, .. + } = child.expect_non_asap() + else { + panic!("expected inner Aggregate{{Rate}}, got {child:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Rate])); + let NonASAPOp::TimeRange { + range, + child: tr_child, + .. + } = child.expect_non_asap() + else { + panic!("expected TimeRange under Rate, got {child:?}"); + }; + assert_eq!(*range, Duration::from_secs(300)); + assert!( + matches!(tr_child.expect_non_asap(), NonASAPOp::Scan { predicates, .. } if predicates.len() == 1) + ); +} + +#[test] +fn histogram_quantile_over_sum_by_le_preserves_grouping() { + // The canonical Prometheus histogram pattern. Previously returned + // UnsupportedFeature because `extract_matrix` couldn't see through the + // `sum by (le)` aggregate; now the `le` grouping survives into the + // canonical tree. + let qe = lower(r#"histogram_quantile(0.99, sum by (le) (rate(http_requests_bucket[5m])))"#); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate{{HistogramQuantile}}, got {qe:?}"); + }; + // The `by (le)` grouping marks the classic cumulative-bucket form. + assert!( + matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if (*q - 0.99).abs() < 1e-9) + ); + // `sum by (le)` survives as a positional Aggregate (by = [2], `le`) over the + // inner Rate — no name-based Partition. + let NonASAPOp::Aggregate { + reduction, + measures, + .. + } = child.expect_non_asap() + else { + panic!("expected `sum by (le)` as a positional Aggregate, got {child:?}"); + }; + assert_eq!(reduction, &Reduction::by(vec![2])); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); +} + +/// The classic `histogram_quantile` aggregate: its `without` keys, `le` +/// column, and output column names. +fn classic_histogram(qe: &OperatorNode) -> (Vec, usize, Vec) { + let NonASAPOp::Aggregate { + reduction: Reduction::Reduce(by), + measures, + .. + } = qe.expect_non_asap() + else { + panic!("expected a reducing Aggregate, got {qe:?}"); + }; + let [AggIntent::HistogramQuantile { le, .. }] = measures.as_slice() else { + panic!("expected HistogramQuantile, got {measures:?}"); + }; + assert!(by.is_without(), "histogram_quantile groups without (le)"); + let names = qe.schema.fields.iter().map(|c| c.name.clone()).collect(); + (by.keys().to_vec(), *le, names) +} + +// A classic histogram_quantile groups `without (le)` and names the child's +// `le` column, even when no matcher or grouping mentions `le`. +#[test] +fn classic_histogram_quantile_groups_without_le() { + let qe = lower("histogram_quantile(0.9, rate(http_duration_seconds_bucket[5m]))"); + let (keys, le, names) = classic_histogram(&qe); + let NonASAPOp::Aggregate { child, .. } = qe.expect_non_asap() else { + unreachable!() + }; + let child = &child.schema; + assert_eq!(child.fields[le].name, "le"); + assert_eq!(keys, vec![le]); + assert_eq!(names, vec!["histogram_quantile"]); +} + +// An explicit `sum by (le, job)` argument keeps `job` and drops `le` and the +// renamed sample value from the output labels. +#[test] +fn classic_histogram_quantile_over_sum_by_keeps_other_labels() { + let qe = + lower("histogram_quantile(0.9, sum by (le, job) (rate(http_duration_seconds_bucket[5m])))"); + let (keys, le, names) = classic_histogram(&qe); + // `sum by (le, job)` outputs `[job, le, sum]`. + assert_eq!((keys, le), (vec![1], 1)); + assert_eq!(names, vec!["job", "histogram_quantile"]); +} + +// Out-of-range and NaN quantiles lower unchanged; execution returns -Inf/+Inf/NaN. +#[test] +fn classic_histogram_quantile_keeps_out_of_range_quantiles() { + for (query, expected) in [ + ("histogram_quantile(-1, x_bucket)", -1.), + ("histogram_quantile(2, x_bucket)", 2.), + ] { + let root = lower(query); + let NonASAPOp::Aggregate { measures, .. } = root.expect_non_asap() else { + panic!("{query}"); + }; + assert!( + matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if *q == expected) + ); + } + let root = lower("histogram_quantile(NaN, x_bucket)"); + let NonASAPOp::Aggregate { measures, .. } = root.expect_non_asap() else { + panic!("NaN"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::HistogramQuantile { q, .. }] if q.is_nan())); +} + +// An argument whose closed output lacks `le` has no buckets. Prometheus +// returns an empty vector; lowering rejects it rather than guess a column. +#[test] +fn classic_histogram_quantile_rejects_an_argument_without_le() { + let error = lower_promql( + "histogram_quantile(0.9, sum by (job) (rate(x_bucket[5m])))", + AccuracyTarget::Exact, + ) + .unwrap_err(); + assert!(error.to_string().contains("le"), "{error}"); +} + +// ── rate / increase carry their own window (no Window node) ───────────────────── + +#[test] +fn rate_has_time_range_child_not_window() { + let qe = lower("rate(http_requests_total[5m])"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected Aggregate for rate, got {qe:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Rate])); + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { + panic!("expected TimeRange child (not Window), got {child:?}"); + }; + assert_eq!(*range, Duration::from_secs(300)); +} + +#[test] +fn increase_maps_to_increase_intent() { + let qe = lower("increase(errors_total[1h])"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected Aggregate for increase, got {qe:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Increase])); + let NonASAPOp::TimeRange { range, .. } = child.expect_non_asap() else { + panic!("expected TimeRange child, got {child:?}"); + }; + assert_eq!(*range, Duration::from_secs(3600)); +} + +// ── outer aggregation over an inner range-vector func is two levels ───────────── + +#[test] +fn sum_over_rate_keeps_both_levels() { + // Regression: `sum(rate(m[w]))` — the most common PromQL shape — must keep + // the cross-series Sum, not collapse to a bare per-series Rate. + let qe = lower("sum(rate(http_requests_total[5m]))"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate{{Sum}}, got {qe:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + let NonASAPOp::Aggregate { + measures, child, .. + } = child.expect_non_asap() + else { + panic!("expected inner Aggregate{{Rate}}, got {child:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Rate])); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); +} + +#[test] +fn sum_by_over_rate_groups_the_outer_sum() { + // `sum by (job) (rate(...))`: the grouping belongs to the OUTER sum and lands + // on a positional `Aggregate.by` (the same shape SQL produces) over the + // label-preserving inner Rate. Leaf = [ts, value, job] → by = [2]. + let qe = lower("sum by (job) (rate(http_requests_total[5m]))"); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate grouped by job, got {qe:?}"); + }; + assert_eq!(reduction, &Reduction::by(vec![2])); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + )); +} + +#[test] +fn count_over_rate_keeps_both_levels() { + // The `Outer::Count` sibling of the `sum(rate(...))` bug. + let qe = lower("count(rate(http_requests_total[5m]))"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate{{Count}}, got {qe:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate]) + )); +} + +#[test] +fn count_over_distinct_over_time_preserves_both_aggregates() { + // One series with window samples [1, 2] produces one distinct-count + // result (value 2). The outer count counts that one series, yielding 1. + for (query, reduction) in [ + ( + "count(distinct_over_time(unique_users[5m]))", + Reduction::by(vec![]), + ), + ( + "count by (job) (distinct_over_time(unique_users[5m]))", + Reduction::by(vec![2]), + ), + ] { + let tree = lower(query); + let NonASAPOp::Aggregate { + measures, + reduction: actual, + child, + .. + } = tree.expect_non_asap() + else { + panic!("expected outer Count: {tree:?}"); + }; + assert!( + matches!(measures.as_slice(), [AggIntent::Count { .. }]), + "{query}: {tree:?}" + ); + assert_eq!(actual, &reduction, "{query}"); + let NonASAPOp::Aggregate { + measures, + reduction, + child, + .. + } = child.expect_non_asap() + else { + panic!("expected inner per-series Cardinality: {tree:?}"); + }; + assert!( + matches!(measures.as_slice(), [AggIntent::Cardinality { .. }]), + "{query}: {tree:?}" + ); + assert_eq!(reduction, &Reduction::PerEntity, "{query}"); + assert!( + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { range, .. } if range.as_secs() == 300) + ); + } +} + +// ── count / cardinality ─────────────────────────────────────────────────────── + +// Both selector fast paths and recursive vector expressions count rows, not values. +#[test] +fn count_never_lowers_to_distinct_sample_values() { + for query in [ + "count(up)", + "count by (job) (up)", + "count without (instance) (up)", + "count(up + 1)", + "count(count_over_time(up[5m]))", + "count_over_time(up[5m])", + ] { + let tree = lower(query); + let intents = all_intents(&tree); + assert!( + intents.iter().any(|i| matches!(i, AggIntent::Count { .. })), + "{query}: {tree:?}" + ); + assert!( + !intents + .iter() + .any(|i| matches!(i, AggIntent::Cardinality { .. })), + "{query}: {tree:?}" + ); + } +} + +#[test] +fn count_over_time_is_count_intent() { + let qe = lower("count_over_time(m[5m])"); + let NonASAPOp::Aggregate { + measures, child, .. + } = qe.expect_non_asap() + else { + panic!("expected Aggregate"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); +} + +#[test] +fn outer_count_counts_series() { + // `count by (symbol) (count_over_time(...))`: inner per-series sample count + // over the window (label-preserving), outer cross-series row count grouped + // on a positional `Aggregate.by`. Leaf = [ts, value, symbol] → symbol = col 2. + let qe = lower("count by (symbol) (count_over_time(financial_last_trade_price[5m]))"); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected outer Aggregate grouped by symbol, got {qe:?}"); + }; + assert_eq!(reduction, &Reduction::by(vec![2])); + assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); + // Inner: Aggregate{Count} over TimeRange (per-series count_over_time). + let NonASAPOp::Aggregate { + measures, child, .. + } = child.expect_non_asap() + else { + panic!("expected Aggregate (count_over_time) under the outer count, got {child:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); +} + +// ── topk / bottomk ──────────────────────────────────────────────────────────── + +#[test] +fn topk_over_count_is_heavy_hitter_topk() { + let qe = lower(r#"topk by (service) (10, count_over_time(requests{env="prod"}[1m]))"#); + // Heavy-hitter: Aggregate{TopK} with grouping resolved to positional ids. + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected Aggregate with TopK, got {qe:?}"); + }; + // `service` is the only group key → resolved to a positional ColumnId. + assert_eq!(reduction.expect_reduce().len(), 1); + assert!(matches!( + measures.as_slice(), + [AggIntent::TopK { k: 10, .. }] + )); + // The count_over_time under the TopK is a TimeRange-backed aggregate. + let NonASAPOp::Aggregate { + measures, child, .. + } = child.expect_non_asap() + else { + panic!("expected Aggregate (count_over_time) under TopK, got {child:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); + let NonASAPOp::TimeRange { range, child, .. } = child.expect_non_asap() else { + panic!("expected TimeRange under Count aggregate, got {child:?}"); + }; + assert_eq!(*range, Duration::from_secs(60)); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); +} + +#[test] +fn topk_over_sum_is_value_weighted_heavy_hitter_topk() { + let qe = lower(r#"topk by (service) (5, sum_over_time(requests{env="prod"}[1m]))"#); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected Aggregate with TopK, got {qe:?}"); + }; + assert_eq!(reduction.expect_reduce().len(), 1); + assert!(matches!( + measures.as_slice(), + [AggIntent::TopK { k: 5, .. }] + )); + let NonASAPOp::Aggregate { + measures, child, .. + } = child.expect_non_asap() + else { + panic!("expected Aggregate (sum_over_time) under TopK, got {child:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); +} + +#[test] +fn topk_over_avg_is_generic_sort_limit() { + let qe = lower("topk by (host) (5, avg_over_time(cpu[5m]))"); + let NonASAPOp::Limit { + n: Some(n), + offset, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected Limit, got {qe:?}"); + }; + assert_eq!(*n, 5); + assert_eq!(*offset, 0); + let NonASAPOp::Sort { + keys, + partition_by, + child, + } = child.expect_non_asap() + else { + panic!("expected Sort under Limit, got {child:?}"); + }; + assert_eq!(keys.len(), 1); + assert!(!keys[0].ascending, "topk ranks descending"); + // `by (host)` is per-group ranking → it rides on `Sort.partition_by` + // (positional), not a `Partition` node (issue #12). `host` is col 2 in + // the per-series avg schema [ts, value, host]. + assert_eq!(partition_by, &vec![2]); + // Underneath: the label-preserving windowed avg aggregate (by: []), no + // intervening Partition. + assert!( + matches!(child.expect_non_asap(), NonASAPOp::Aggregate { reduction, measures, .. } + if reduction == &Reduction::PerEntity && matches!(measures.as_slice(), [AggIntent::Avg { .. }])), + "expected bare per-series Avg aggregate under Sort, got {child:?}" + ); +} + +#[test] +fn ungrouped_topk_over_sum_is_heavy_hitter() { + let qe = lower("topk(5, sum_over_time(m[5m]))"); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Aggregate { .. })); + assert!(has_intent(&qe, |i| matches!(i, AggIntent::Sum { .. }))); + assert!(has_intent(&qe, |i| matches!( + i, + AggIntent::TopK { k: 5, .. } + ))); +} + +#[test] +fn bottomk_over_count_is_generic_sort_ascending() { + // `bottomk` is never a heavy-hitter (descending=false), even over count. + let qe = lower("bottomk(3, count_over_time(m[5m]))"); + let NonASAPOp::Limit { + n: Some(n), child, .. + } = qe.expect_non_asap() + else { + panic!("expected Limit, got {qe:?}"); + }; + assert_eq!(*n, 3); + let NonASAPOp::Sort { keys, .. } = child.expect_non_asap() else { + panic!("expected Sort"); + }; + assert!(keys[0].ascending, "bottomk ranks ascending"); + // Count intent is still present (as the inner aggregate), no TopK. + assert!(has_intent(&qe, |i| matches!(i, AggIntent::Count { .. }))); + assert!(!has_intent(&qe, |i| matches!(i, AggIntent::TopK { .. }))); +} + +#[test] +fn bottomk_is_always_generic_sort_ascending() { + let qe = lower("bottomk(3, count_over_time(m[5m]))"); + let NonASAPOp::Limit { + n: Some(n), child, .. + } = qe.expect_non_asap() + else { + panic!("expected Limit, got {qe:?}"); + }; + assert_eq!(*n, 3); + let NonASAPOp::Sort { keys, .. } = child.expect_non_asap() else { + panic!("expected Sort"); + }; + assert!(keys[0].ascending, "bottomk ranks ascending"); +} + +#[test] +fn topk_count_output_schema_carries_group_key() { + // The inner Count is per-series (label-preserving), so the group-by key + // (`service`) flows through to the outer TopK's `by` column. Leaf schema = + // [ts, value, service] → TopK groups on service (col 2). + let qe = lower("topk by (service) (5, count_over_time(m[1m]))"); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected Aggregate{{TopK}}, got {qe:?}"); + }; + assert_eq!( + reduction, + &Reduction::by(vec![2]), + "service is col 2 in [ts, value, service]" + ); + assert!(matches!( + measures.as_slice(), + [AggIntent::TopK { k: 5, .. }] + )); + // Inner Count aggregate is visible with its TimeRange child. + let NonASAPOp::Aggregate { + measures, child, .. + } = child.expect_non_asap() + else { + panic!("expected inner Aggregate{{Count}}, got {child:?}"); + }; + assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); + assert!(matches!( + child.expect_non_asap(), + NonASAPOp::TimeRange { .. } + )); +} + +// ── binary ops ──────────────────────────────────────────────────────────────── + +#[test] +fn binary_op_division() { + let qe = lower("rate(a[5m]) / rate(b[5m])"); + let NonASAPOp::BinaryOp { + operator: BinaryOperator { kind: op, .. }, + lhs, + rhs, + .. + } = qe.expect_non_asap() + else { + panic!("expected BinaryOp, got {qe:?}"); + }; + assert_eq!(*op, BinaryOpKind::Arithmetic(ArithmeticOpKind::Div)); + assert!( + matches!(lhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) + ); + assert!( + matches!(rhs.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } if matches!(measures.as_slice(), [AggIntent::Rate])) + ); +} + +#[test] +fn binary_op_with_on_grouping() { + let qe = lower("a / on(host) b"); + let NonASAPOp::BinaryOp { + operator: BinaryOperator { vector_match, .. }, + .. + } = qe.expect_non_asap() + else { + panic!("expected BinaryOp, got {qe:?}"); + }; + let vm = vector_match.as_ref().expect("vector_match present"); + use asap_types::pre_asap::VectorMatchKind; + assert_eq!(vm.kind, VectorMatchKind::On); + assert_eq!(vm.labels, vec!["host".to_string()]); +} + +// `bool` changes a comparison from a filter to a 0/1 result, so the IR must +// carry it. +#[test] +fn bool_comparisons_are_distinct() { + let op = |q: &str| match lower(q).expect_non_asap() { + NonASAPOp::BinaryOp { + operator, + return_bool, + .. + } => (operator.kind.clone(), *return_bool), + NonASAPOp::Filter { + pred: asap_types::ir::Predicate(ScalarExpr::Compare { op, .. }), + .. + } => (BinaryOpKind::Compare(op.clone()), false), + NonASAPOp::Project { cols, .. } => { + let ScalarExpr::Case { branches, .. } = &cols[1].expr else { + panic!() + }; + let ScalarExpr::Compare { op, .. } = &branches[0].0 else { + panic!() + }; + (BinaryOpKind::Compare(op.clone()), true) + } + other => panic!("expected BinaryOp, got {other:?}"), + }; + assert_eq!( + op("a > 1"), + (BinaryOpKind::Compare(CompareOpKind::Gt), false) + ); + assert_eq!( + op("a > bool 1"), + (BinaryOpKind::Compare(CompareOpKind::Gt), true) + ); + assert_eq!( + op("a == bool on(job) b"), + (BinaryOpKind::Compare(CompareOpKind::Eq), true) + ); +} + +#[test] +fn binary_op_binds_each_branch_against_its_own_schema() { + // Each side scans a different metric and groups by a different label. With a + // single root schema threaded to both branches, the left scan would leak the + // right's group key (and vice-versa). Per-branch binding keeps them separate. + let qe = lower("count by (job) (a) / count by (region) (b)"); + let NonASAPOp::BinaryOp { lhs, rhs, .. } = qe.expect_non_asap() else { + panic!("expected BinaryOp, got {qe:?}"); + }; + let lcols = scan_columns(lhs); + let rcols = scan_columns(rhs); + assert!( + lcols.iter().any(|c| c == "job") && !lcols.iter().any(|c| c == "region"), + "lhs scan schema leaked the rhs key: {lcols:?}" + ); + assert!( + rcols.iter().any(|c| c == "region") && !rcols.iter().any(|c| c == "job"), + "rhs scan schema leaked the lhs key: {rcols:?}" + ); +} + +/// Collect every `AggIntent` in the tree, root-to-leaf. +fn all_intents(e: &OperatorNode) -> Vec { + let mut out = Vec::new(); + collect_intents(e, &mut out); + out +} + +fn collect_intents(e: &OperatorNode, out: &mut Vec) { + match e.expect_non_asap() { + NonASAPOp::Aggregate { + measures, child, .. + } => { + out.extend(measures.iter().cloned()); + collect_intents(child, out); + } + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => collect_intents(child, out), + NonASAPOp::BinaryOp { lhs, rhs, .. } => { + collect_intents(lhs, out); + collect_intents(rhs, out); + } + _ => {} + } +} + +/// True if any `AggIntent` anywhere in the tree satisfies `pred`. +fn has_intent bool>(e: &OperatorNode, pred: F) -> bool { + all_intents(e).iter().any(pred) +} + +/// Field names on the first `Scan` reachable by descending single-child nodes. +fn scan_columns(e: &OperatorNode) -> Vec { + match e.expect_non_asap() { + NonASAPOp::Scan { schema, .. } => schema.fields.iter().map(|c| c.name.clone()).collect(), + NonASAPOp::Aggregate { child, .. } + | NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Filter { child, .. } + | NonASAPOp::Sort { child, .. } + | NonASAPOp::Limit { child, .. } => scan_columns(child), + _ => vec![], + } +} + +// ── without(...) grouping (issue #39) ─────────────────────────────────────────── + +#[test] +fn without_grouping_lowers_to_the_exclusion_form() { + // `sum without (instance) (rate(m[5m]))` — a cross-series reduction over the + // per-series rate, grouped by every label except `instance`. The excluded + // label is stored positionally (the SchemaResolver seeds it), the grouping is the + // `without` form, and the output schema stays open. + let qe = lower("sum without (instance) (rate(m[5m]))"); + let NonASAPOp::Aggregate { + reduction, + measures, + child, + .. + } = qe.expect_non_asap() + else { + panic!("expected an Aggregate, got {qe:?}"); + }; + let by = reduction.expect_reduce(); + assert!(by.is_without()); + assert_eq!(by.keys().len(), 1, "excluded `instance`"); + assert!(matches!(measures.as_slice(), [AggIntent::Sum { .. }])); + // The inner per-series rate is preserved (label-preserving) under the outer + // cross-series `without` reduction. + assert!( + matches!(child.expect_non_asap(), NonASAPOp::Aggregate { measures, .. } + if matches!(measures.as_slice(), [AggIntent::Rate])) + ); + assert!(!qe.schema.clone().closed); +} + +// ── parameter validation (reject rather than silently truncate/garble) ────────── + +#[test] +fn fractional_or_negative_topk_k_is_rejected() { + // `as u64` would silently truncate 2.7→2 / saturate -1→0. + assert!(lower_promql("topk(2.7, count_over_time(m[1m]))", AccuracyTarget::Exact).is_err()); + assert!(lower_promql("bottomk(2.5, sum_over_time(m[1m]))", AccuracyTarget::Exact).is_err()); +} + +#[test] +fn out_of_range_quantile_phi_is_accepted() { + // Prometheus defines out-of-range phi results; lowering must preserve it. + for query in [ + "quantile(1.5, up)", + "quantile_over_time(1.5, m[5m])", + "histogram_quantile(2.0, rate(b_bucket[5m]))", + ] { + assert!( + lower_promql(query, AccuracyTarget::Exact).is_ok(), + "{query}" + ); + } +} + +#[test] +fn function_wrapped_range_vector_is_rejected_not_stripped() { + // `rate(abs(m[5m]))` must NOT silently lower as `rate(m[5m])` — the wrapper + // is rejected (here, at parse or in extract_matrix), never stripped. + assert!( + lower_promql("rate(abs(http_requests_total[5m]))", AccuracyTarget::Exact).is_err(), + "function-wrapped range vector should be rejected" + ); +} + +#[test] +fn pathologically_nested_query_is_rejected_not_stack_overflow() { + // 300 nested parens parse fine but exceed the walker's depth limit (256); + // it must return an error, not overflow the stack. + let q = format!("{}m{}", "(".repeat(300), ")".repeat(300)); + let err = lower_promql(&q, AccuracyTarget::Exact).unwrap_err(); + assert!(format!("{err}").contains("nesting"), "got {err}"); +} + +// Behavior: every parser-accepted `fill` modifier form is rejected with a +// fill-specific lowering error rather than silently dropped. +#[test] +fn fill_modifiers_are_rejected_not_ignored() { + for q in [ + "a + fill(0) b", + "a + fill_left(1) b", + "a + fill_right(2) b", + "a + fill_left(1) fill_right(2) b", + "a + fill_right(2) fill_left(1) b", + "a + on(job) fill(0) b", + "a * ignoring(instance) group_left(env) fill_right(0) b", + "a > bool on(job) fill(0) b", + "sum(a - on(job) group_right fill_left(0) b)", + ] { + match lower_promql(q, AccuracyTarget::Exact) { + Err(LoweringError::UnsupportedFeature(m)) if m.contains("`fill`") => {} + other => panic!("expected fill rejection for {q:?}, got {other:?}"), + } + } +} + +// ── accuracy propagation ────────────────────────────────────────────────────── + +#[test] +fn accuracy_target_flows_into_quantile_intent() { + let qe = lower_promql( + "quantile_over_time(0.9, m[5m])", + AccuracyTarget::Epsilon(0.01), + ) + .unwrap(); + let NonASAPOp::Aggregate { measures, .. } = qe.expect_non_asap() else { + panic!("expected Aggregate"); + }; + assert!(matches!( + &measures[0], + AggIntent::Quantile { accuracy: AccuracyTarget::Epsilon(e), .. } if (*e - 0.01).abs() < 1e-12 + )); +} + +// ── schema flow (positional, carried on Scan; derived on demand) ───────────────── + +#[test] +fn aggregate_output_schema_preserves_time_axis_and_labels() { + let qe = lower(r#"quantile_over_time(0.99, http_request_duration{env="prod"}[5m])"#); + // Per-series reduction: the root is Aggregate { TimeRange { Scan } }. + // The SchemaResolver adds all referenced label names (group keys AND filter + // predicate columns) to the scan schema, so `env` appears as a column + // even though it is only used as a filter. + // per_series_reduction_schema preserves the time axis and all label columns. + let NonASAPOp::Aggregate { .. } = qe.expect_non_asap() else { + panic!("expected Aggregate, got {qe:?}"); + }; + let schema = &qe.schema; + let names: Vec<&str> = schema.fields.iter().map(|c| c.name.as_str()).collect(); + assert_eq!(names, vec!["ts", "value", "env"]); + assert_eq!( + schema.time_index, + Some(0), + "per-series over_time preserves the time axis" + ); +} + +#[test] +fn scan_schema_carries_ts_value_and_group_keys() { + // `service` is a group key → the SchemaResolver lands it in the self-contained + // Scan schema (positional). `env` is only a filter, so it is not a column. + let qe = lower("count by (service) (count_over_time(requests[1m]))"); + fn find_scan(n: &OperatorNode) -> &OperatorNode { + match n.expect_non_asap() { + NonASAPOp::Scan { .. } => n, + NonASAPOp::TimeRange { child, .. } + | NonASAPOp::Aggregate { child, .. } + | NonASAPOp::Filter { child, .. } => find_scan(child), + other => panic!("unexpected node {other:?}"), + } + } + let NonASAPOp::Scan { schema, .. } = find_scan(&qe).expect_non_asap() else { + unreachable!() + }; + let mut names: Vec<&str> = schema.fields.iter().map(|c| c.name.as_str()).collect(); + names.sort(); + assert_eq!(names, vec!["service", "ts", "value"]); + assert_eq!(schema.time_index, Some(0)); // ts +} + +// ── batch entry point ───────────────────────────────────────────────────────── + +#[test] +fn batch_lowers_each_entry_and_reads_per_query_accuracy() { + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![ + BatchEntry { + query: Query("rate(a[5m])".into()), + requirements: QueryRequirements::default(), + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }, + BatchEntry { + query: Query("quantile_over_time(0.9, b[5m])".into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(AccuracyTarget::Epsilon(0.02)), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }, + ]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(1_000)), + ..Default::default() + }, + ..Default::default() + }), + }; + let results = lower_promql_workload(&workload, 0).expect("valid workload"); + assert_eq!(results.len(), 2); +} + +#[test] +fn batch_rejects_non_promql_language() { + use asap_types::workload::SqlDialect; + let workload = PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::SQL(SqlDialect::DataFusionSQL), + query_batch: Some(vec![BatchEntry { + query: Query("SELECT 1".into()), + requirements: QueryRequirements::default(), + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: None, + }; + assert!(matches!( + lower_promql_workload(&workload, 0), + Err(LoweringError::WrongLanguage(_)) + )); +} + +// ── #12: one home per grouping concept (the canonical `Partition` node is removed) ── +// +// `Partition` and `Aggregate.by` were two ways to express grouping. #12 collapses +// them: a reducing GROUP BY → `Aggregate.by`; per-group *ranking* (split without +// reduce) → `Sort.partition_by`; parallel sharding → a deployment's own +// physical stage. There is no longer a canonical `Partition` node. These +// tests pin both surviving canonical homes. + +#[test] +fn reducing_group_by_lowers_to_aggregate_by() { + // Cross-series reduce, no keys → bare `Aggregate { reduction: Reduce([]) }`. + let q = lower("sum(http_requests_total)"); + assert!( + matches!(q.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } if reduction == &Reduction::by(vec![])) + ); + + // Cross-series reduce grouped by a label → `Aggregate.reduction`. + let q = lower("sum by (job) (http_requests_total)"); + assert!( + matches!(q.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } + if reduction.expect_reduce().len() == 1) + ); + + // Reduce over a label-preserving `rate` grouped by a label → still + // `Aggregate.reduction` (the keys resolve against rate's preserved schema). + let q = lower("sum by (job) (rate(http_requests_total[5m]))"); + assert!( + matches!(q.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } + if reduction.expect_reduce().len() == 1) + ); +} + +#[test] +fn generic_topk_grouping_lowers_to_sort_partition_by() { + // Per-group ranking (`topk by (host)`, non-heavy-hitter) groups *without* + // reducing → the grouping rides on `Sort.partition_by`, and the windowed + // reduction beneath stays label-preserving (`by: []`). No `Partition` node. + let q = lower("topk by (host) (5, avg_over_time(cpu[5m]))"); + let NonASAPOp::Limit { child, .. } = q.expect_non_asap() else { + panic!("expected Limit, got {q:?}"); + }; + let NonASAPOp::Sort { + partition_by, + child, + .. + } = child.expect_non_asap() + else { + panic!("expected Sort, got {child:?}"); + }; + assert_eq!(partition_by, &vec![2], "host is col 2 in [ts, value, host]"); + assert!( + matches!(child.expect_non_asap(), NonASAPOp::Aggregate { reduction, .. } if reduction == &Reduction::PerEntity) + ); +} + +#[test] +fn topk_over_bare_selector_by_label_ranks_per_group() { + // `topk(3, http_requests_total) by (job)` — top-3 series per `job`. A bare + // instant selector ranks its OWN samples; it must not be wrapped in an + // implicit cross-series `Sum`, which would collapse the `job` partition + // label before `Sort.partition_by` resolves it (issue #30 — follow-up to the + // Partition→Sort.partition_by reframe in #12). Expected: + // Limit{3} → Sort{value desc, partition_by:[job]} → Scan + let q = lower("topk(3, http_requests_total) by (job)"); + let NonASAPOp::Limit { + n: Some(n), child, .. + } = q.expect_non_asap() + else { + panic!("expected Limit, got {q:?}"); + }; + assert_eq!(*n, 3); + let NonASAPOp::Sort { + keys, + partition_by, + child, + } = child.expect_non_asap() + else { + panic!("expected Sort, got {child:?}"); + }; + assert!(!keys[0].ascending, "topk ranks descending"); + assert_eq!(partition_by, &vec![2], "job is col 2 in [ts, value, job]"); + // No implicit reducing aggregate — the selector is label-preserving, so the + // sort is directly over the selector horizon (the `job` label survives to partition by). + assert!( + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })), + "ranking is over the bare selector horizon, not a reducing Aggregate, got {child:?}" + ); + assert!( + !has_intent(&q, |i| matches!(i, AggIntent::Sum { .. })), + "no implicit Sum is introduced over a bare selector" + ); +} + +#[test] +fn topk_over_bare_selector_ranks_raw_samples() { + // Even without `by`, `topk(3, m)` ranks the raw instant-vector samples — it + // does not sum them. The sort sits directly over the Scan, partition empty. + let q = lower("topk(3, http_requests_total)"); + let NonASAPOp::Limit { child, .. } = q.expect_non_asap() else { + panic!("expected Limit, got {q:?}"); + }; + let NonASAPOp::Sort { + partition_by, + child, + .. + } = child.expect_non_asap() + else { + panic!("expected Sort, got {child:?}"); + }; + assert!(partition_by.is_empty(), "no `by` → global ranking"); + assert!( + matches!(child.expect_non_asap(), NonASAPOp::TimeRange { child, .. } if matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })) + ); + assert!(!has_intent(&q, |i| matches!(i, AggIntent::Sum { .. }))); +} + +// ── Issue #109: histogram_quantiles fans out into one branch per φ ────────── + +/// The `(label value, intent)` of each `histogram_quantiles` branch. +fn quantile_branches(q: &OperatorNode) -> Vec<(String, AggIntent)> { + let NonASAPOp::Concat { children, .. } = q.expect_non_asap() else { + panic!("expected a Concat at the root, got {q:?}"); + }; + children + .iter() + .map(|c| { + let NonASAPOp::PromqlRelabel { value, child, .. } = c.expect_non_asap() else { + panic!("expected PromqlRelabel per branch, got {c:?}"); + }; + let ScalarExpr::Literal(ScalarValue::Utf8(v)) = value else { + panic!("expected a literal label value, got {value:?}"); + }; + let NonASAPOp::Aggregate { measures, .. } = child.expect_non_asap() else { + panic!("expected an Aggregate under the PromqlRelabel, got {child:?}"); + }; + (v.clone(), measures[0].clone()) + }) + .collect() +} + +#[test] +fn histogram_quantiles_rejects_unrepresented_native_histograms() { + assert!(lower_promql( + r#"histogram_quantiles(testhistogram3, "q", 0, 0.25, 1)"#, + AccuracyTarget::Exact + ) + .is_err()); +} + +#[test] +fn histogram_quantiles_over_classic_buckets_interpolates() { + // `_bucket` argument → exact cumulative-bucket interpolation, never a sketch. + let q = lower(r#"histogram_quantiles(request_duration_seconds_bucket, "q", 0.5, 0.9)"#); + for (_, intent) in quantile_branches(&q) { + assert!( + matches!(intent, AggIntent::HistogramQuantile { .. }), + "classic buckets → HistogramQuantile, got {intent:?}" + ); + } +} + +#[test] +fn histogram_quantiles_branches_are_union_compatible() { + // `Concat` derives its schema from the first child, so every branch must + // agree on column names — the φ lives in the label, not the column name. + let q = lower(r#"histogram_quantiles(testhistogram3_bucket, "q", 0.5, 0.9)"#); + let NonASAPOp::Concat { children, .. } = q.expect_non_asap() else { + panic!("expected Concat"); + }; + let shapes: Vec> = children + .iter() + .map(|c| { + c.schema + .clone() + .fields + .iter() + .map(|c| c.name.clone()) + .collect() + }) + .collect(); + assert_eq!(shapes[0], shapes[1], "branches must be union-compatible"); + assert_eq!(shapes[0], vec!["value".to_string(), "q".to_string()]); + assert_eq!( + q.schema.fields.len(), + 2, + "the merged schema describes every branch" + ); +} + +#[test] +fn histogram_quantiles_uses_the_given_label_name() { + let q = lower(r#"histogram_quantiles(h_bucket, "phi", 0.5)"#); + let NonASAPOp::Concat { children, .. } = q.expect_non_asap() else { + panic!("expected Concat"); + }; + let NonASAPOp::PromqlRelabel { dst, .. } = children[0].expect_non_asap() else { + panic!("expected PromqlRelabel"); + }; + assert_eq!(dst, "phi"); +} + +#[test] +fn histogram_quantiles_formats_small_quantiles_like_prometheus() { + // `labels.FormatOpenMetricsFloat`: Go's %g, so exponent form below 1e-4. + let q = lower(r#"histogram_quantiles(h_bucket, "q", 0.00001)"#); + assert_eq!(quantile_branches(&q)[0].0, "1e-05"); +} + +#[test] +fn histogram_quantiles_rejects_an_out_of_range_quantile() { + // Same rule as `histogram_quantile(φ, …)` — one bad φ fails the whole call. + for q in [ + r#"histogram_quantiles(h_bucket, "q", -0.1)"#, + r#"histogram_quantiles(h_bucket, "q", 1.01)"#, + r#"histogram_quantiles(h_bucket, "q", 0.5, NaN)"#, + ] { + assert!( + lower_promql(q, AccuracyTarget::Exact).is_err(), + "{q} should be rejected" + ); + } +} + +// ── TimeRange.kind: instant vs range selectors ────────────────────────────────── + +#[test] +fn bare_instant_selector_is_an_instant_time_range() { + // `up` reads the latest sample per series within the workload's ingestion + // interval (1s in `support::workload`): an `Instant` lookback of that length. + let qe = lower("up"); + let NonASAPOp::TimeRange { range, kind, child } = qe.expect_non_asap() else { + panic!("expected TimeRange, got {qe:?}"); + }; + assert_eq!(*kind, TimeRangeKind::Instant); + assert_eq!(*range, Duration::from_secs(1)); + assert!(matches!(child.expect_non_asap(), NonASAPOp::Scan { .. })); +} + +#[test] +fn explicit_range_selector_is_a_range_time_range() { + // `m[5m]` keeps its own window and is a `Range` selection — both under a + // range function and as a bare matrix selector. + let qe = lower("rate(m[5m])"); + let NonASAPOp::Aggregate { child, .. } = qe.expect_non_asap() else { + panic!("expected Aggregate, got {qe:?}"); + }; + let NonASAPOp::TimeRange { range, kind, .. } = child.expect_non_asap() else { + panic!("expected TimeRange, got {child:?}"); + }; + assert_eq!(*kind, TimeRangeKind::Range); + assert_eq!(*range, Duration::from_secs(300)); + + let qe = lower("m[5m]"); + assert!(matches!( + qe.expect_non_asap(), + NonASAPOp::TimeRange { + kind: TimeRangeKind::Range, + .. + } + )); +} + +#[test] +fn instant_and_range_selectors_of_equal_length_stay_distinct() { + // The kind is part of the shape: a 1s range selector is not the same tree as + // the 1s instant lookback injected around a bare selector. + assert_ne!(lower("up"), lower("up[1s]")); +} + +// ── the `bool` modifier → `return_bool` ───────────────────────────────────────── + +#[test] +fn vector_scalar_comparison_without_bool_filters() { + let qe = lower("up > 0"); + assert!(matches!(qe.expect_non_asap(), NonASAPOp::Filter { .. })); + assert!(matches!( + support::sample_expression(&qe), + ScalarExpr::Compare { + op: CompareOpKind::Gt, + .. + } + )); +} + +#[test] +fn vector_scalar_comparison_with_bool_sets_return_bool() { + let qe = lower("up > bool 0"); + assert!(matches!( + support::sample_expression(&qe), + ScalarExpr::Case { .. } + )); + assert_ne!(qe, lower("up > 0")); +} + +#[test] +fn vector_vector_comparison_with_bool_sets_return_bool() { + // `a > bool b` — the modifier lands on the vector/vector op itself, with + // the default (ignoring nothing) match. + let qe = lower("a > bool b"); + let NonASAPOp::BinaryOp { + operator, + return_bool, + lhs, + rhs, + } = qe.expect_non_asap() + else { + panic!("expected BinaryOp, got {qe:?}"); + }; + assert!(*return_bool); + assert_eq!(operator.kind, BinaryOpKind::Compare(CompareOpKind::Gt)); + assert!(matches!(lhs.expect_non_asap(), NonASAPOp::TimeRange { .. })); + assert!(matches!(rhs.expect_non_asap(), NonASAPOp::TimeRange { .. })); + assert!(!lower("a > b").expect_non_asap().children().is_empty()); + assert_ne!(qe, lower("a > b")); +} + +#[test] +fn bool_modifier_composes_with_vector_matching() { + let qe = lower("a > bool on(job) b"); + let NonASAPOp::BinaryOp { + operator, + return_bool, + .. + } = qe.expect_non_asap() + else { + panic!("expected BinaryOp, got {qe:?}"); + }; + assert!(*return_bool); + let vm = operator.vector_match.as_ref().expect("on(job) present"); + assert_eq!(vm.labels, vec!["job".to_string()]); +} + +// ── scalar expressions: negation, arithmetic, comparison ──────────────────────── + +#[test] +fn scalar_negation_of_time_is_a_negative_expression() { + // `-time()` is a scalar expression; its negation stays structural (the + // operand is not a constant to fold) and follows PromQL numeric rules. + let qe = support::scalar_root("-time()"); + let ScalarExpr::Negative { expr, semantics } = &qe else { + panic!("expected ScalarExpr(Negative), got {qe:?}"); + }; + assert_eq!(*semantics, ExprSemantics::Promql); + assert!(matches!(expr.as_ref(), ScalarExpr::EvalTimestamp)); + // Scalar-shaped: no time index. +} + +#[test] +fn scalar_negation_of_a_constant_still_folds() { + // `-(2)` is constant: it folds to one literal rather than a `Negative`. + assert_eq!( + support::promql_scalar(&support::scalar_root("-(2)")), + Some(-2.0) + ); +} + +#[test] +fn scalar_arithmetic_carries_promql_semantics() { + let qe = support::scalar_root("time() - 1"); + let ScalarExpr::Arithmetic { + op, + left, + right, + semantics, + } = &qe + else { + panic!("expected scalar(Arithmetic), got {qe:?}"); + }; + assert_eq!(*op, ArithmeticOpKind::Sub); + assert_eq!(*semantics, ExprSemantics::Promql); + assert!(matches!(left.as_ref(), ScalarExpr::EvalTimestamp)); + assert!(matches!( + right.as_ref(), + ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0 + )); +} + +#[test] +fn scalar_bool_comparison_is_a_zero_one_case_with_promql_semantics() { + // `1 < bool 2` → `Case(Compare(1 < 2) → 1.0, else 0.0)`: PromQL yields 0/1. + let qe = support::scalar_root("1 < bool 2"); + let ScalarExpr::Case { + operand, + branches, + else_expr, + } = &qe + else { + panic!("expected scalar(Case), got {qe:?}"); + }; + assert!(operand.is_none()); + let [(when, then)] = branches.as_slice() else { + panic!("expected one branch, got {branches:?}"); + }; + let ScalarExpr::Compare { + left, + op, + right, + semantics, + } = when + else { + panic!("expected a Compare condition, got {when:?}"); + }; + assert_eq!(*op, CompareOpKind::Lt); + assert_eq!(*semantics, ExprSemantics::Promql); + assert!(matches!(left.as_ref(), ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); + assert!(matches!(right.as_ref(), ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 2.0)); + assert!(matches!(then, ScalarExpr::Literal(ScalarValue::Float64(v)) if *v == 1.0)); + assert!(matches!( + else_expr.as_deref(), + Some(ScalarExpr::Literal(ScalarValue::Float64(v))) if *v == 0.0 + )); +} + +#[test] +fn scalar_comparison_without_bool_is_rejected() { + // PromQL has no scalar filter: a scalar/scalar comparison needs `bool`. + for q in ["1 < 2", "time() > 0", "(1 + 1) == 2"] { + assert!( + lower_promql(q, AccuracyTarget::Exact).is_err(), + "{q} must be rejected without `bool`" + ); + } +} + +#[test] +fn label_matcher_predicates_carry_promql_semantics() { + let qe = lower(r#"up{job="api"}"#); + let NonASAPOp::TimeRange { child, .. } = qe.expect_non_asap() else { + panic!("expected TimeRange, got {qe:?}"); + }; + let NonASAPOp::Scan { predicates, .. } = child.expect_non_asap() else { + panic!("expected Scan, got {child:?}"); + }; + assert!(matches!( + &predicates[0].0, + ScalarExpr::Compare { + semantics: ExprSemantics::Promql, + .. + } + )); +} diff --git a/crates/frontend-promql/tests/unified_scalar_design.rs b/crates/frontend-promql/tests/unified_scalar_design.rs new file mode 100644 index 000000000..4e3f3063e --- /dev/null +++ b/crates/frontend-promql/tests/unified_scalar_design.rs @@ -0,0 +1,129 @@ +//! Scalar expressions never become constant-wrapper operators. +#[path = "unified_support.rs"] +mod support; +use asap_types::ir::{NonASAPOp, QueryRoot, ScalarExpr}; +use asap_types::pre_asap::{ArithmeticOpKind, ScalarValue}; +use asap_types::types::AccuracyTarget; + +fn root(query: &str) -> QueryRoot { + asap_frontend_promql::unified::lower_promql_query_workload( + &support::workload(query, AccuracyTarget::Exact), + 0, + ) + .unwrap() + .remove(0) +} + +#[test] +fn standalone_scalars_are_expressions() { + for query in [ + "2", + "time()", + "scalar(sum(up)) + 1", + "1 < bool 2", + "-time()", + ] { + let QueryRoot::Scalar(expr) = root(query) else { + panic!("{query} became an operator") + }; + expr.scalar_type(&Default::default()).unwrap(); + } +} + +#[test] +fn arithmetic_projects_the_sample_and_preserves_full_identity_and_time() { + let QueryRoot::Operator(node) = root("up * 2") else { + panic!() + }; + let NonASAPOp::Project { child, cols, .. } = node.expect_non_asap() else { + panic!() + }; + assert!(child.schema.has_promql_series_identity()); + assert!(node.schema.has_promql_series_identity()); + assert_eq!(node.schema.time_index, child.schema.time_index); + let value = node.schema.column_id("value").unwrap(); + assert!( + matches!(&cols[value].expr, ScalarExpr::Arithmetic { op: ArithmeticOpKind::Mul, right, .. } if **right == ScalarExpr::Literal(ScalarValue::Float64(2.0))) + ); + assert!(cols.iter().any(|c| matches!(&c.expr, ScalarExpr::FunctionCall { name, .. } if name == "promql_drop_metric_name"))); +} + +#[test] +fn non_bool_comparisons_keep_vector_samples_even_with_scalar_on_left() { + for query in ["up > 0", "0 < up"] { + let QueryRoot::Operator(node) = root(query) else { + panic!() + }; + let NonASAPOp::Filter { child, .. } = node.expect_non_asap() else { + panic!() + }; + assert_eq!(node.schema, child.schema); + } +} + +#[test] +fn bool_comparison_projects_zero_or_one() { + let QueryRoot::Operator(node) = root("up > bool 0") else { + panic!() + }; + let NonASAPOp::Project { cols, .. } = node.expect_non_asap() else { + panic!() + }; + assert!(matches!( + cols[node.schema.column_id("value").unwrap()].expr, + ScalarExpr::Case { .. } + )); +} + +#[test] +fn scalar_plan_dependencies_remain_visible() { + let QueryRoot::Operator(node) = root("up * scalar(sum(up))") else { + panic!() + }; + assert_eq!(node.children().len(), 2); +} + +/// Pointwise functions own scalar parameters, including vector-to-scalar reads. +#[test] +fn pointwise_functions_are_typed_scalar_projections() { + for query in [ + "abs(up)", + "round(up, scalar(sum(other)))", + "clamp(up, time() - 1, time())", + "year(up)", + "hour()", + ] { + let QueryRoot::Operator(node) = root(query) else { + panic!() + }; + let NonASAPOp::Project { cols, .. } = node.expect_non_asap() else { + panic!("{query}: expected projection") + }; + assert!(matches!( + &cols[node.schema.column_id("value").unwrap()].expr, + ScalarExpr::FunctionCall { .. } + )); + node.validate_structure().unwrap(); + } +} + +/// Negation preserves the metric name and complete identity unlike multiplication. +#[test] +fn unary_minus_preserves_identity() { + let QueryRoot::Operator(node) = root("-up") else { + panic!() + }; + let NonASAPOp::Project { child, cols, .. } = node.expect_non_asap() else { + panic!() + }; + assert_eq!(node.schema, child.schema); + assert!(matches!( + &cols[node.schema.column_id("value").unwrap()].expr, + ScalarExpr::Negative { .. } + )); + for (index, col) in cols.iter().enumerate() { + if index != node.schema.column_id("value").unwrap() { + assert_eq!(col.expr, ScalarExpr::Column(index)); + } + } +} diff --git a/crates/frontend-promql/tests/unified_support.rs b/crates/frontend-promql/tests/unified_support.rs new file mode 100644 index 000000000..26ff6e802 --- /dev/null +++ b/crates/frontend-promql/tests/unified_support.rs @@ -0,0 +1,101 @@ +use std::rc::Rc; + +use asap_frontend_promql::unified::{ + lower_promql_workload, lower_promql_workload_with_histograms, HistogramCatalog, PromqlError, +}; +use asap_types::ir::{NonASAPOp, OperatorNode, ScalarExpr}; +use asap_types::pre_asap::ScalarValue; +use asap_types::types::AccuracyTarget; +use asap_types::workload::{ + AccuracyRequirement, BatchEntry, DataWorkload, DurationMs, Evidence, PlanningWorkload, + Predictability, Query, QueryLanguage, QueryRequirements, QueryWorkload, TimeSelection, +}; + +pub fn workload(query: &str, accuracy: AccuracyTarget) -> PlanningWorkload { + PlanningWorkload { + query_workload: QueryWorkload { + language: QueryLanguage::PromQL, + query_batch: Some(vec![BatchEntry { + query: Query(query.into()), + requirements: QueryRequirements { + accuracy: AccuracyRequirement::Explicit(accuracy), + ..Default::default() + }, + predictability: Predictability::Unknown, + invocations: 1, + execute_at: None, + time_selection: TimeSelection::default(), + }]), + repeating_queries: None, + }, + data_workload: Some(DataWorkload { + data_ingestion_interval: Evidence { + value: Some(DurationMs(1_000)), + ..Default::default() + }, + ..Default::default() + }), + } +} + +#[allow(dead_code)] +pub fn lower_promql( + query: &str, + accuracy: AccuracyTarget, +) -> Result, PromqlError> { + let mut lowered = lower_promql_workload(&workload(query, accuracy), 0)?; + Ok(lowered.remove(0)) +} + +#[allow(dead_code)] +pub fn lower_promql_with_histograms( + query: &str, + accuracy: AccuracyTarget, + histograms: HistogramCatalog, +) -> Result, PromqlError> { + let mut lowered = + lower_promql_workload_with_histograms(&workload(query, accuracy), histograms, 0)?; + Ok(lowered.remove(0)) +} + +/// The value of a bare PromQL numeric literal / folded constant at an +/// scalar position (`Literal(Float64(v))`); `None` for any +/// other shape. +#[allow(dead_code)] +pub fn promql_scalar(node: &ScalarExpr) -> Option { + match node { + ScalarExpr::Literal(ScalarValue::Float64(v)) => Some(*v), + _ => None, + } +} + +/// Export the logical graph before physical materialization assigns timing. +#[allow(dead_code)] +pub fn logical_asap_dag(root: &Rc) -> asap_types::ir::export::LogicalASAPDAG { + asap_types::ir::export::compile_logical_asap_dag(root).expect("logical ASAP DAG export") +} + +#[allow(dead_code)] +pub fn scalar_root(query: &str) -> ScalarExpr { + match asap_frontend_promql::unified::lower_promql_query_workload( + &workload(query, AccuracyTarget::Exact), + 0, + ) + .unwrap() + .remove(0) + { + asap_types::ir::QueryRoot::Scalar(expr) => expr, + _ => panic!("expected scalar root: {query}"), + } +} + +#[allow(dead_code)] +pub fn sample_expression(node: &OperatorNode) -> &ScalarExpr { + match node.expect_non_asap() { + NonASAPOp::Project { cols, .. } => { + &cols[node.schema.column_id("value").unwrap_or(cols.len() - 1)].expr + } + NonASAPOp::Filter { pred, .. } => &pred.0, + other => panic!("expected sample expression, got {other:?}"), + } +}