diff --git a/AGENTS.md b/AGENTS.md index fdfedd82f..3bad13863 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -2,7 +2,7 @@ - Work in a dedicated Git worktree unless the user explicitly says otherwise. - Create new worktrees from `origin/main` unless another base is specified. - Fetch the base first, and never disturb an existing dirty working tree. + Fetch the base first, and never disturb an existing dirty working DAG. - Keep communication and generated prose concise. - Prefer the minimally complex implementation that satisfies the requirement. - Do not introduce a conceptual layer, abstraction, or public interface without diff --git a/crates/asap-aware-mapping/src/accuracy/allocation.rs b/crates/asap-aware-mapping/src/accuracy/allocation.rs index e06761ec6..77849d9c8 100644 --- a/crates/asap-aware-mapping/src/accuracy/allocation.rs +++ b/crates/asap-aware-mapping/src/accuracy/allocation.rs @@ -23,7 +23,7 @@ pub struct AccuracyAllocation { impl AccuracyAllocation { /// The end-to-end budget left for everything below `layers[0]` — what - /// the inner subtree must satisfy as a whole (it re-splits internally). + /// the inner sub-DAG must satisfy as a whole (it re-splits internally). /// `None` for a single-layer allocation. pub fn inner_target(&self, shape: &CompositionShape) -> Option { let inner = &self.layers[1..]; diff --git a/crates/asap-aware-mapping/src/accuracy/composition.rs b/crates/asap-aware-mapping/src/accuracy/composition.rs index f3355b281..f42e8f158 100644 --- a/crates/asap-aware-mapping/src/accuracy/composition.rs +++ b/crates/asap-aware-mapping/src/accuracy/composition.rs @@ -25,7 +25,7 @@ impl DefaultAccuracyModel { /// `(1 + ε_total) = Π (1 + ε_i)` ⇒ for two factors /// `ε_in + ε_out + ε_in·ε_out`; written out as the sum of all - /// cross-products so the expression tree is exact for any input count. + /// cross-products so the expression DAG is exact for any input count. fn multiplicative( op: &CompositionOperator, inputs: &[ResultGuarantee], diff --git a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs index 46cbc06f5..15f8f27df 100644 --- a/crates/asap-aware-mapping/src/accuracy/reconciliation.rs +++ b/crates/asap-aware-mapping/src/accuracy/reconciliation.rs @@ -3,20 +3,20 @@ //! //! ## The gap this closes //! -//! `asap_types::pre_asap::cse::share_common_subtrees` (pre-ASAP CSE) only -//! ever merges two subtrees that are *exactly* [`PartialEq`]-equal, +//! `asap_types::pre_asap::cse::share_common_sub_dags` (pre-ASAP CSE) only +//! ever merges two sub-DAGs that are *exactly* [`PartialEq`]-equal, //! including their [`AggIntent`]'s `accuracy: AccuracyTarget` field. Two //! otherwise-identical aggregates that differ *only* in how tight an //! accuracy bound they ask for — `quantile(0.99, x)` at `epsilon=0.01` for //! one consumer, the same `quantile(0.99, x)` at `epsilon=0.05` for //! another — are therefore never the same `Rc`, never collapse into one -//! [`crate::replacement::TargetSubDAGCandidates`], and [`crate::replacement::SharedSubtreeStrategy`] +//! [`crate::replacement::TargetSubDAGCandidates`], and [`crate::replacement::SharedSubDAGStrategy`] //! never even gets a `TargetSubDAG` with `consumer_count >= 2` to propose //! sharing for. This crate would build two entirely independent sketches //! for what is conceptually one computation, even though a single sketch //! built to the tighter of the two bounds would answer both. //! -//! This module is **additive**, not a relaxation of `share_common_subtrees` +//! This module is **additive**, not a relaxation of `share_common_sub_dags` //! itself: `accuracy` still participates in exact structural equality //! everywhere else in this crate (correctness elsewhere — e.g. a downstream //! consumer that pattern-matches on a specific `AccuracyTarget` — depends on @@ -24,7 +24,7 @@ //! enough to share" that sits entirely inside the [`ReplacementStrategy`] //! extension point: one more candidate a [`crate::cost_model::CostModel`] //! may or may not prefer, never a forced rewrite and never a change to what -//! `share_common_subtrees` itself merges. +//! `share_common_sub_dags` itself merges. //! //! ## What counts as a "near-duplicate", and why //! @@ -39,7 +39,7 @@ //! intent has no `AccuracyTarget` to reconcile in the first place. //! 2. Same `reduction` (grouping), same `output_names`, and the same shared //! `child` (`Rc::ptr_eq`, or value-equal for two independently-built but -//! identical subtrees CSE conservatively declined to alias) — the same +//! identical sub-DAGs CSE conservatively declined to alias) — the same //! "identical everything else" bar [`crate::rollup::RollupStrategy`] and //! [`crate::topk_reuse::TopKLimitReuseStrategy`] already hold their own //! sibling-reuse candidates to. @@ -50,7 +50,7 @@ //! trivially "always tightest"). //! 5. The tighter candidate's own **output** schema carries a provable //! unique key (`Schema::has_unique_key`) — the exact legality gate -//! `share_common_subtrees` itself applies (see `cse.rs`'s "Legality" +//! `share_common_sub_dags` itself applies (see `cse.rs`'s "Legality" //! section) and [`crate::rollup::RollupStrategy::is_legal_rollup_source`] //! already reuses verbatim for the identical reason: a producer's output //! is only safely reusable across a second, independent consumer when @@ -120,12 +120,12 @@ //! tag), because it needs its own cost treatment in //! [`crate::cost_model::DefaultCostModel::estimate_cost`], not just its own //! label. Every other `Replacement::Rewrite` shape that reaches -//! `estimate_cost` (`SharedSubtreeStrategy`'s `CseRecompute`, `Rollup`'s and +//! `estimate_cost` (`SharedSubDAGStrategy`'s `CseRecompute`, `Rollup`'s and //! `TopKLimitReuse`'s `LogicalRewrite`) really does rebuild `target` from a //! different source, so pricing it as "one `cse_recompute_cost` of `target` //! itself, per consumer" is the right shape of cost. This strategy's //! candidate never rebuilds `target` at all — it reads `rc` (the tighter -//! sibling), which — per this module's own safety argument — is a subtree +//! sibling), which — per this module's own safety argument — is a sub-DAG //! this crate is already going to build regardless of whether `target` //! reads from it too. Pricing it with the same "rebuild `target`, once per //! consumer" formula would charge it for work it never does, and — because @@ -137,7 +137,7 @@ //! pin against. `estimate_cost` instead prices this shape as a //! [`crate::cost_model::CostModel::cse_shared_maintenance_cost`] read //! against `rc`'s **own** bound summary — the same order-of-magnitude, -//! per-family cost `SharedSubtreeStrategy`'s own `CseShare` candidate is +//! per-family cost `SharedSubDAGStrategy`'s own `CseShare` candidate is //! priced with, reflecting "one more reference into a structure that's //! already being maintained" rather than "build a whole new one." //! @@ -146,7 +146,7 @@ //! that sibling group propagate the uses through its selected implementation. //! Accuracy edges are directed strictly from looser to tighter budgets, so //! they cannot cycle among themselves; both near-duplicates also have the -//! same structural child, so adding the edge preserves the reference graph's +//! same structural child, so adding the edge preserves the reference DAG's //! parent-before-child topological ordering. use std::cmp::Ordering; @@ -301,7 +301,7 @@ impl AccuracyReconciliationStrategy { /// /// Also requires the candidate's own *output* schema to carry a provable /// unique key ([`Schema::has_unique_key`]) — the exact legality gate - /// `pre_asap::cse::share_common_subtrees` already applies to its own + /// `pre_asap::cse::share_common_sub_dags` already applies to its own /// sharing decisions, and [`crate::rollup::RollupStrategy`] already /// reuses verbatim for the identical reason (see that module's /// `is_legal_rollup_source` doc, point 4): a producer's output is only @@ -392,12 +392,12 @@ mod tests { use super::*; use crate::cost_model::{CostModel, DefaultCostModel}; use asap_types::post_asap::SketchAlgorithm; - use asap_types::pre_asap::cse::share_common_subtrees; + use asap_types::pre_asap::cse::share_common_sub_dags; use asap_types::pre_asap::query_expr::{GroupKeys, Source}; use asap_types::pre_asap::schema::{Column, ColumnId, DataType, Schema}; /// `[ts(0), value(1), job(2)]`. - /// A unique-keyed scan (`[ts]`) so `share_common_subtrees` is actually + /// A unique-keyed scan (`[ts]`) so `share_common_sub_dags` is actually /// willing to hoist it — see `Schema::has_unique_key`/`cse.rs`'s own /// "Legality" section: a producer with no provable unique key is always /// inserted fresh, never hoisted, regardless of structural equality. @@ -658,24 +658,24 @@ mod tests { assert!(!strategy.matches(&TargetSubDAG::new(&exact))); } - // ── exact structural equality / share_common_subtrees is unchanged ──── + // ── exact structural equality / share_common_sub_dags is unchanged ──── #[test] - fn share_common_subtrees_still_never_merges_differing_accuracy() { + fn share_common_sub_dags_still_never_merges_differing_accuracy() { // The additive guarantee this issue explicitly must not violate: // pre-ASAP CSE's own exact-equality merge stays exact. Two // aggregates differing only in `accuracy` must come back as two // distinct `Rc`s, not one shared `Rc` — AccuracyReconciliationStrategy // is the *only* place cross-accuracy sharing gets proposed, never - // `share_common_subtrees` itself. + // `share_common_sub_dags` itself. let scan = metric_scan(); let a = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); let b = (*quantile(0.99, AccuracyTarget::Epsilon(0.05), &scan)).clone(); - let roots = share_common_subtrees(vec![("a", a), ("b", b)]); + let roots = share_common_sub_dags(vec![("a", a), ("b", b)]); assert!( !Rc::ptr_eq(&roots[0].1, &roots[1].1), - "share_common_subtrees must not merge aggregates with different AccuracyTarget" + "share_common_sub_dags must not merge aggregates with different AccuracyTarget" ); assert_ne!( roots[0].1, roots[1].1, @@ -696,14 +696,14 @@ mod tests { #[test] fn identical_accuracy_still_merges_via_ordinary_cse() { // Sanity check the fixture itself: truly identical aggregates - // (same accuracy too) still merge via share_common_subtrees's own + // (same accuracy too) still merge via share_common_sub_dags's own // exact equality — unrelated to this module, but pins the contrast // with the test above. let scan = metric_scan(); let a = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); let b = (*quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan)).clone(); - let roots = share_common_subtrees(vec![("a", a), ("b", b)]); + let roots = share_common_sub_dags(vec![("a", a), ("b", b)]); assert!(Rc::ptr_eq(&roots[0].1, &roots[1].1)); } @@ -714,7 +714,7 @@ mod tests { // `by(vec![])` (global aggregation) reports no unique key // (`aggregate_output_schema`'s own `unique_keys = if by.is_empty() .. // { vec![] } ..`) — the same legality gate - // `share_common_subtrees`/`RollupStrategy` apply, which this + // `share_common_sub_dags`/`RollupStrategy` apply, which this // strategy must not bypass (module docs, point 5). let scan = metric_scan(); let tight = global_quantile(0.99, AccuracyTarget::Epsilon(0.01), &scan); @@ -848,7 +848,7 @@ mod tests { // independently-built but structurally identical loose queries // merge onto one Rc via ordinary CSE), *and* a separate, // single-consumer tight sibling exists over the same input — the - // scenario the issue itself targets: `SharedSubtreeStrategy`'s own + // scenario the issue itself targets: `SharedSubDAGStrategy`'s own // CseShare/CseRecompute pair is on the table for the loose target's // own 2 consumers at the same time as this strategy's "read the // tight sibling instead" candidate. diff --git a/crates/asap-aware-mapping/src/analytical_cost.rs b/crates/asap-aware-mapping/src/analytical_cost.rs index ed2a316db..3c19960a9 100644 --- a/crates/asap-aware-mapping/src/analytical_cost.rs +++ b/crates/asap-aware-mapping/src/analytical_cost.rs @@ -378,7 +378,7 @@ pub enum HashJoinBuildSide { /// transient edge buffering, and state retained across the horizon are /// separate values. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub struct PhysicalDagNode { +pub struct PhysicalDAGNode { pub id: String, pub operator: PhysicalOperator, pub children: Vec, @@ -410,8 +410,8 @@ pub struct PhysicalNodeEvidence { /// lowering. Keeping the root and evidence beside the nodes prevents callers /// from estimating a valid node list with a different entry point or snapshot. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] -pub struct EvidenceBackedPhysicalDag { - pub nodes: Vec, +pub struct EvidenceBackedPhysicalDAG { + pub nodes: Vec, pub root: String, pub evidence: HashMap, } @@ -422,7 +422,7 @@ impl OperatorStatisticsProvider for HashMap { .get(node_id) .ok_or_else(|| AnalyticalCostError::MissingOperatorStatistics(node_id.into()))?; if evidence.physical_id != node_id { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "evidence map key differs from embedded physical identity", )); } @@ -430,22 +430,22 @@ impl OperatorStatisticsProvider for HashMap { } } -impl OperatorStatisticsProvider for EvidenceBackedPhysicalDag { +impl OperatorStatisticsProvider for EvidenceBackedPhysicalDAG { fn statistics(&self, node_id: &str) -> Result { let evidence = self .evidence .get(node_id) .ok_or_else(|| AnalyticalCostError::MissingOperatorStatistics(node_id.into()))?; let node = self.nodes.iter().find(|node| node.id == node_id).ok_or( - AnalyticalCostError::InvalidPhysicalDag("evidence has no matching physical node"), + AnalyticalCostError::InvalidPhysicalDAG("evidence has no matching physical node"), )?; if evidence.physical_id != node_id { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "evidence map key differs from embedded physical identity", )); } if node.output_buffer_bytes != evidence.output_buffer_bytes { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "physical node buffer differs from evidence snapshot", )); } @@ -462,8 +462,8 @@ pub enum ExecutionMultiplicity { /// Borrowed inputs for one physical-DAG estimate. This remains available /// independently for diagnostics; plan selection should use /// [`estimate_physical_dag_comparison`] so scope equality is mandatory. -pub struct PhysicalDagEstimateRequest<'a> { - pub nodes: &'a [PhysicalDagNode], +pub struct PhysicalDAGEstimateRequest<'a> { + pub nodes: &'a [PhysicalDAGNode], pub root: &'a str, pub scope: &'a ComparisonScope, pub statistics: &'a dyn OperatorStatisticsProvider, @@ -471,7 +471,7 @@ pub struct PhysicalDagEstimateRequest<'a> { } #[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)] -pub struct PhysicalDagComparisonEstimate { +pub struct PhysicalDAGComparisonEstimate { pub raw: ResourceEstimate, pub candidate: ResourceEstimate, } @@ -479,16 +479,16 @@ pub struct PhysicalDagComparisonEstimate { /// Estimate two plans only after proving that their source, snapshot, /// predicate, event-time, recurrence, and horizon scopes are identical. pub fn estimate_physical_dag_comparison( - raw: PhysicalDagEstimateRequest<'_>, - candidate: PhysicalDagEstimateRequest<'_>, -) -> Result { + raw: PhysicalDAGEstimateRequest<'_>, + candidate: PhysicalDAGEstimateRequest<'_>, +) -> Result { validate_comparison_scopes(raw.scope, candidate.scope)?; if raw.cache_profile != candidate.cache_profile { return Err(AnalyticalCostError::ComparisonScopeMismatch( "cache profile", )); } - Ok(PhysicalDagComparisonEstimate { + Ok(PhysicalDAGComparisonEstimate { raw: estimate_physical_dag_with_cache( raw.nodes, raw.root, @@ -510,7 +510,7 @@ pub fn estimate_physical_dag_comparison( /// are additive; peak memory is simulated over a child-before-parent schedule /// and releases transient child outputs after their last consumer. pub fn estimate_physical_dag( - nodes: &[PhysicalDagNode], + nodes: &[PhysicalDAGNode], root: &str, scope: &ComparisonScope, statistics: &(impl OperatorStatisticsProvider + ?Sized), @@ -519,7 +519,7 @@ pub fn estimate_physical_dag( } pub fn estimate_physical_dag_with_cache( - nodes: &[PhysicalDagNode], + nodes: &[PhysicalDAGNode], root: &str, scope: &ComparisonScope, statistics: &(impl OperatorStatisticsProvider + ?Sized), @@ -527,16 +527,16 @@ pub fn estimate_physical_dag_with_cache( ) -> Result { let evaluation_count = scope.validate()?; let cache = resolve_cache_profile(cache_profile, evaluation_count, scope.data_arrival)?; - let by_id: HashMap<&str, &PhysicalDagNode> = nodes.iter().map(|n| (n.id.as_str(), n)).collect(); + let by_id: HashMap<&str, &PhysicalDAGNode> = nodes.iter().map(|n| (n.id.as_str(), n)).collect(); if by_id.len() != nodes.len() { - return Err(AnalyticalCostError::InvalidPhysicalDag("duplicate node id")); + return Err(AnalyticalCostError::InvalidPhysicalDAG("duplicate node id")); } let mut visiting = HashSet::new(); let mut visited = HashSet::new(); let mut order = Vec::new(); fn visit<'a>( id: &'a str, - nodes: &HashMap<&'a str, &'a PhysicalDagNode>, + nodes: &HashMap<&'a str, &'a PhysicalDAGNode>, visiting: &mut HashSet<&'a str>, visited: &mut HashSet<&'a str>, order: &mut Vec<&'a str>, @@ -545,11 +545,11 @@ pub fn estimate_physical_dag_with_cache( return Ok(()); } if !visiting.insert(id) { - return Err(AnalyticalCostError::InvalidPhysicalDag("cycle")); + return Err(AnalyticalCostError::InvalidPhysicalDAG("cycle")); } let node = nodes .get(id) - .ok_or(AnalyticalCostError::InvalidPhysicalDag("missing node"))?; + .ok_or(AnalyticalCostError::InvalidPhysicalDAG("missing node"))?; for child in &node.children { visit(child, nodes, visiting, visited, order)?; } @@ -586,7 +586,7 @@ pub fn estimate_physical_dag_with_cache( } } _ if node.source_coverage.is_some() => { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "only scan nodes may declare source coverage", )); } @@ -595,7 +595,7 @@ pub fn estimate_physical_dag_with_cache( validate_operator_statistics(node, node_statistics, &by_id, &resolved_statistics)?; if node.retained_bytes > 0 && matches!(node.execution, ExecutionMultiplicity::PerEvaluation) { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "per-evaluation node cannot retain state across the horizon", )); } @@ -604,7 +604,7 @@ pub fn estimate_physical_dag_with_cache( if matches!(node.execution, ExecutionMultiplicity::Once) && matches!(child.execution, ExecutionMultiplicity::PerEvaluation) { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "build-once node cannot consume a per-evaluation child", )); } @@ -612,7 +612,7 @@ pub fn estimate_physical_dag_with_cache( && matches!(child.execution, ExecutionMultiplicity::Once) && child.retained_bytes == 0 { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "per-evaluation node reads a non-retained build-once child", )); } @@ -623,7 +623,7 @@ pub fn estimate_physical_dag_with_cache( .iter() .any(|expected| !consumed_sources.contains(&expected)) { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "physical scans omit a comparison-scope source", )); } @@ -699,7 +699,7 @@ pub fn estimate_physical_dag_with_cache( } for child in &node.children { let remaining = remaining_consumers.get_mut(child.as_str()).ok_or( - AnalyticalCostError::InvalidPhysicalDag("invalid consumer count"), + AnalyticalCostError::InvalidPhysicalDAG("invalid consumer count"), )?; *remaining -= 1; if *remaining == 0 { @@ -722,9 +722,9 @@ pub fn estimate_physical_dag_with_cache( } fn validate_operator_statistics( - node: &PhysicalDagNode, + node: &PhysicalDAGNode, node_statistics: &OperatorStatistics, - nodes: &HashMap<&str, &PhysicalDagNode>, + nodes: &HashMap<&str, &PhysicalDAGNode>, statistics: &HashMap<&str, OperatorStatistics>, ) -> Result<(), AnalyticalCostError> { let arity = expected_input_arity(node.operator, node.children.len()); @@ -735,7 +735,7 @@ fn validate_operator_statistics( }); } if node.children.len() != arity.dag_children { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "operator child count does not match physical arity", )); } @@ -759,7 +759,7 @@ fn validate_operator_statistics( for (input_index, child_id) in node.children.iter().enumerate() { let child = nodes .get(child_id.as_str()) - .ok_or(AnalyticalCostError::InvalidPhysicalDag("missing node"))?; + .ok_or(AnalyticalCostError::InvalidPhysicalDAG("missing node"))?; let child_statistics = &statistics[child.id.as_str()]; if node_statistics.input(input_index) != Some(child_statistics.output()) { return Err(AnalyticalCostError::ConflictingEdgeStatistics { @@ -775,7 +775,7 @@ fn validate_operator_statistics( } fn validate_promql_child_edges( - node: &PhysicalDagNode, + node: &PhysicalDAGNode, node_statistics: &OperatorStatistics, statistics: &HashMap<&str, OperatorStatistics>, ) -> Result<(), AnalyticalCostError> { @@ -2218,7 +2218,7 @@ pub enum AnalyticalCostError { input_index: usize, }, #[error("invalid physical DAG: {0}")] - InvalidPhysicalDag(&'static str), + InvalidPhysicalDAG(&'static str), } fn checked_bytes(parts: &[u64]) -> Result { @@ -2471,7 +2471,7 @@ mod tests { }; let promql = promql_edge(1, 10, PromqlValueKind::Vector); let nodes = vec![ - PhysicalDagNode { + PhysicalDAGNode { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -2480,7 +2480,7 @@ mod tests { retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, }, - PhysicalDagNode { + PhysicalDAGNode { id: "filter".into(), operator: filter_operator(), children: vec!["scan".into()], @@ -2810,7 +2810,7 @@ mod tests { fn physical_dag_counts_shared_scan_once_and_uses_live_memory() { let coverage = comparison_scope().sources[0].clone(); let nodes = vec![ - PhysicalDagNode { + PhysicalDAGNode { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -2819,7 +2819,7 @@ mod tests { retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, }, - PhysicalDagNode { + PhysicalDAGNode { id: "left".into(), operator: filter_operator(), children: vec!["scan".into()], @@ -2828,7 +2828,7 @@ mod tests { retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, }, - PhysicalDagNode { + PhysicalDAGNode { id: "right".into(), operator: filter_operator(), children: vec!["scan".into()], @@ -2837,7 +2837,7 @@ mod tests { retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, }, - PhysicalDagNode { + PhysicalDAGNode { id: "root".into(), operator: PhysicalOperator::Concat, children: vec!["left".into(), "right".into()], @@ -2901,7 +2901,7 @@ mod tests { fn physical_dag_separates_build_once_from_per_evaluation_work() { let coverage = comparison_scope().sources[0].clone(); let nodes = vec![ - PhysicalDagNode { + PhysicalDAGNode { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -2910,7 +2910,7 @@ mod tests { retained_bytes: 0, execution: ExecutionMultiplicity::Once, }, - PhysicalDagNode { + PhysicalDAGNode { id: "state".into(), operator: aggregate_operator(), children: vec!["scan".into()], @@ -2919,7 +2919,7 @@ mod tests { retained_bytes: 32, execution: ExecutionMultiplicity::Once, }, - PhysicalDagNode { + PhysicalDAGNode { id: "read".into(), operator: PhysicalOperator::Limit { limit: 1, @@ -3055,7 +3055,7 @@ mod tests { let coverage = comparison_scope().sources[0].clone(); let nodes = vec![ - PhysicalDagNode { + PhysicalDAGNode { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -3064,7 +3064,7 @@ mod tests { retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, }, - PhysicalDagNode { + PhysicalDAGNode { id: "filter".into(), operator: filter_operator(), children: vec!["scan".into()], @@ -3130,7 +3130,7 @@ mod tests { let coverage = comparison_scope().sources[0].clone(); let nodes = vec![ - PhysicalDagNode { + PhysicalDAGNode { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -3139,7 +3139,7 @@ mod tests { retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, }, - PhysicalDagNode { + PhysicalDAGNode { id: "filter".into(), operator: filter_operator(), children: vec!["scan".into()], @@ -3193,7 +3193,7 @@ mod tests { fn physical_dag_accepts_an_empty_operator_output() { let coverage = comparison_scope().sources[0].clone(); let nodes = vec![ - PhysicalDagNode { + PhysicalDAGNode { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -3202,7 +3202,7 @@ mod tests { retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, }, - PhysicalDagNode { + PhysicalDAGNode { id: "filter".into(), operator: filter_operator(), children: vec!["scan".into()], @@ -3241,7 +3241,7 @@ mod tests { #[test] fn physical_dag_rejects_a_scan_not_covered_by_its_scope() { - let nodes = vec![PhysicalDagNode { + let nodes = vec![PhysicalDAGNode { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -3279,7 +3279,7 @@ mod tests { #[test] fn physical_dag_rejects_a_scan_without_explicit_coverage() { - let nodes = vec![PhysicalDagNode { + let nodes = vec![PhysicalDAGNode { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -3320,7 +3320,7 @@ mod tests { predicates: vec![], info_matchers: vec![], }); - let nodes = vec![PhysicalDagNode { + let nodes = vec![PhysicalDAGNode { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -3343,7 +3343,7 @@ mod tests { assert_eq!( estimate_physical_dag(&nodes, "scan", &scope, &provided), - Err(AnalyticalCostError::InvalidPhysicalDag( + Err(AnalyticalCostError::InvalidPhysicalDAG( "physical scans omit a comparison-scope source" )) ); @@ -3353,7 +3353,7 @@ mod tests { fn build_once_parent_cannot_consume_a_per_evaluation_child() { let coverage = comparison_scope().sources[0].clone(); let nodes = vec![ - PhysicalDagNode { + PhysicalDAGNode { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -3362,7 +3362,7 @@ mod tests { retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, }, - PhysicalDagNode { + PhysicalDAGNode { id: "aggregate".into(), operator: aggregate_operator(), children: vec!["scan".into()], @@ -3398,7 +3398,7 @@ mod tests { assert_eq!( estimate_physical_dag(&nodes, "aggregate", &comparison_scope(), &provided), - Err(AnalyticalCostError::InvalidPhysicalDag( + Err(AnalyticalCostError::InvalidPhysicalDAG( "build-once node cannot consume a per-evaluation child" )) ); @@ -3407,7 +3407,7 @@ mod tests { #[test] fn scoped_comparison_rejects_different_source_snapshots() { let coverage = comparison_scope().sources[0].clone(); - let nodes = vec![PhysicalDagNode { + let nodes = vec![PhysicalDAGNode { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -3434,14 +3434,14 @@ mod tests { assert_eq!( estimate_physical_dag_comparison( - PhysicalDagEstimateRequest { + PhysicalDAGEstimateRequest { nodes: &nodes, root: "scan", scope: &raw_scope, statistics: &provided, cache_profile: &no_cache, }, - PhysicalDagEstimateRequest { + PhysicalDAGEstimateRequest { nodes: &nodes, root: "scan", scope: &candidate_scope, @@ -3632,13 +3632,13 @@ mod tests { ); } - fn cache_test_scan() -> (Vec, HashMap) { + fn cache_test_scan() -> (Vec, HashMap) { let edge = EdgeStatistics { rows: 100, bytes: 1_000, }; ( - vec![PhysicalDagNode { + vec![PhysicalDAGNode { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], diff --git a/crates/asap-aware-mapping/src/cost_model.rs b/crates/asap-aware-mapping/src/cost_model.rs index 0bf7c0d80..6b3c32e1e 100644 --- a/crates/asap-aware-mapping/src/cost_model.rs +++ b/crates/asap-aware-mapping/src/cost_model.rs @@ -35,8 +35,8 @@ //! ## CSE sharing (issue #237, #223 stage 4) //! //! [`CseCandidate`]/[`ShareDecision`]/[`CostModel::cse_share_decision`] below -//! decide whether a CSE-detected shared subtree -//! ([`asap_types::pre_asap::cse::share_common_subtrees`], issue #223 stages +//! decide whether a CSE-detected shared sub-DAG +//! ([`asap_types::pre_asap::cse::share_common_sub_dags`], issue #223 stages //! 1-2, PR #235) is actually worth sharing, via a real Volcano/Cascades-style //! cost comparison rather than a fixed rule. See //! `docs/design_docs/cse-cost-model-decision.md` for the full design discussion (why @@ -266,24 +266,24 @@ fn finite_rate(units_per_second: f64) -> Option { .then_some(CostRate(units_per_second)) } -/// A CSE-detected, legality-gated shared subtree with two or more consumers +/// A CSE-detected, legality-gated shared sub-DAG with two or more consumers /// — the unit [`CostModel::cse_share_decision`] decides over. Built by /// [`CandidateLogicalASAPDAGs::cost_sorted`](crate::replacement::CandidateLogicalASAPDAGs::cost_sorted) /// (via [`crate::replacement`]'s own `cse_preference`) the first time it -/// needs a representative bound node for a subtree that -/// [`asap_types::pre_asap::cse::share_common_subtrees`] already collapsed +/// needs a representative bound node for a sub-DAG that +/// [`asap_types::pre_asap::cse::share_common_sub_dags`] already collapsed /// onto one `Rc` for two or more workload roots. See /// `docs/design_docs/cse-cost-model-decision.md`. pub struct CseCandidate<'a> { - /// The shared pre-ASAP subtree itself. - pub subtree: &'a QueryExpr, - /// The `SummaryNode` this subtree bound to — gives the cost model the + /// The shared pre-ASAP sub-DAG itself. + pub sub_dag: &'a QueryExpr, + /// The `SummaryNode` this sub-DAG bound to — gives the cost model the /// concrete `SummaryFamilyType`/`(kind, params)` actually at stake, not /// just the pre-ASAP shape. pub bound_summary: &'a SummaryNode, - /// How many workload roots reference this exact shared subtree, counted + /// How many workload roots reference this exact shared sub-DAG, counted /// once up front over the whole workload (always >= 2 — a candidate is - /// only ever constructed for an actually-shared subtree). + /// only ever constructed for an actually-shared sub-DAG). pub consumer_count: usize, } @@ -367,23 +367,23 @@ pub enum ShareDecision { } /// Default [`CostModel::cse_recompute_cost`]: a structural-size proxy — the -/// number of *unique* nodes in `subtree`'s DAG +/// number of *unique* nodes in `sub_dag`'s DAG /// ([`asap_types::pre_asap::cse::dag_node_count`], the same module this /// candidate's sharing was detected in). Deliberately **not** a raw -/// `serde_json` serialization length: after CSE, `subtree` is generally a -/// DAG, not a tree (a `CseCandidate` only exists because something got +/// `serde_json` serialization length: after CSE, `sub_dag` generally has +/// internal sharing (a `CseCandidate` only exists because something got /// shared), and a naive full serialization re-serializes — over-counts — -/// any descendant `subtree` already shares internally, once per parent +/// any descendant `sub_dag` already shares internally, once per parent /// that references it, instead of once for the whole DAG. `dag_node_count` /// dedupes by `Rc` pointer identity, so it charges each unique node's /// contribution exactly once regardless of how many places within -/// `subtree` reference it. Cheap to compute (one pass, no serialization), +/// `sub_dag` reference it. Cheap to compute (one pass, no serialization), /// and still scales with real structural complexity — a genuinely tiny -/// leaf costs little to recompute, a deep multi-join subtree costs a lot. +/// leaf costs little to recompute, a deep multi-join sub-DAG costs a lot. /// A deployment with real per-row/per-update cost knowledge should /// override [`CostModel::cse_recompute_cost`] instead of relying on this. -pub fn default_cse_recompute_cost(subtree: &QueryExpr) -> Cost { - Cost(asap_types::pre_asap::cse::dag_node_count(subtree) as f64) +pub fn default_cse_recompute_cost(sub_dag: &QueryExpr) -> Cost { + Cost(asap_types::pre_asap::cse::dag_node_count(sub_dag) as f64) } /// Default [`CostModel::cse_shared_maintenance_cost`]: a small @@ -567,12 +567,12 @@ pub trait CostModel { ) } - /// Estimate the one-time cost of recomputing `candidate.subtree` + /// Estimate the one-time cost of recomputing `candidate.sub_dag` /// independently at a single use site. Default: /// [`default_cse_recompute_cost`] (a structural-size proxy). See /// `docs/design_docs/cse-cost-model-decision.md`. fn cse_recompute_cost(&self, candidate: &CseCandidate) -> Cost { - default_cse_recompute_cost(candidate.subtree) + default_cse_recompute_cost(candidate.sub_dag) } /// Estimate the cost of maintaining `candidate.bound_summary` as one @@ -666,7 +666,7 @@ pub trait CostModel { Cost(1.0) } - /// Cost of recomputing `candidate.subtree` once, from the pre-ASAP/raw + /// Cost of recomputing `candidate.sub_dag` once, from the pre-ASAP/raw /// path. Units: cost units per recomputation — the `raw_recompute_cost` /// term of `recompute_cost_rate`. Default: delegates to /// [`cse_recompute_cost`](Self::cse_recompute_cost) (the same @@ -740,14 +740,14 @@ pub trait CostModel { /// already belongs to [`rank_candidates`](Self::rank_candidates) (for a /// [`SketchAlgorithmStrategy`](crate::replacement::SketchAlgorithmStrategy) /// group) and [`cse_share_decision`](Self::cse_share_decision) (for a - /// [`SharedSubtreeStrategy`](crate::replacement::SharedSubtreeStrategy) + /// [`SharedSubDAGStrategy`](crate::replacement::SharedSubDAGStrategy) /// group). /// /// One method covers both candidate shapes this crate ships: /// `candidate.replacement`'s [`Replacement::Summary`] arm (a /// `SketchAlgorithmStrategy` candidate — the bound `SummaryNode` is right /// there, nothing to reconstruct) and its [`Replacement::Rewrite`] arm - /// (a `SharedSubtreeStrategy` share-vs-recompute candidate — no bound + /// (a `SharedSubDAGStrategy` share-vs-recompute candidate — no bound /// `SummaryNode` of its own, since sharing is a decision about a target /// already bound some other way; a representative binding is recovered /// from `target` itself). `target` is threaded through explicitly @@ -1044,7 +1044,7 @@ impl CostModel for DefaultCostModel { /// `cse_share_decision` already compares against each other. `NaN` /// only if `target` itself can't be bound at all (schema derivation /// failed) — never expected for a target that's already part of a - /// legitimate workload tree. + /// legitimate workload DAG. /// /// **Exception**: a [`ReplacementProvenance::AccuracyReconciliation`] /// candidate (issue #273) never rebuilds `target` — it reads a @@ -1062,7 +1062,7 @@ impl CostModel for DefaultCostModel { match &candidate.replacement { Replacement::Summary(node) => { let cse = CseCandidate { - subtree: target.root, + sub_dag: target.root, bound_summary: node, consumer_count, }; @@ -1075,7 +1075,7 @@ impl CostModel for DefaultCostModel { return f64::NAN; }; let cse = CseCandidate { - subtree: rc, + sub_dag: rc, bound_summary: &sibling_bound, // One additional reference into `rc`'s own (already // necessary) build, from this one consumer's @@ -1092,7 +1092,7 @@ impl CostModel for DefaultCostModel { return f64::NAN; }; let cse = CseCandidate { - subtree: target.root, + sub_dag: target.root, bound_summary: &bound, consumer_count, }; @@ -1410,11 +1410,11 @@ mod tests { assert!(default_cse_recompute_cost(&nested) > default_cse_recompute_cost(&leaf)); } - /// The DAG-awareness this proxy exists for: a subtree that internally + /// The DAG-awareness this proxy exists for: a sub-DAG that internally /// re-references one shared descendant (e.g. after single-query CSE, /// `x op x` collapsing both branches onto one `Rc`) must cost the same /// as if that descendant only appeared once — not double, the way a - /// naive tree-shaped size measure (a full serialization, or an + /// naive per-path size measure (a full serialization, or an /// identity-blind recursive walk) would count it. #[test] fn default_recompute_cost_does_not_double_count_an_internally_shared_descendant() { @@ -1448,7 +1448,7 @@ mod tests { default_cse_recompute_cost(&with_sharing), Cost(2.0), "internal sharing: Join + 1 shared Scan (referenced twice) = \ - 2 unique nodes, not 3 — a tree-shaped size measure would \ + 2 unique nodes, not 3 — a per-path size measure would \ wrongly charge for the shared Scan twice" ); } @@ -1473,7 +1473,7 @@ mod tests { #[test] fn cse_share_decision_shares_when_recompute_dominates_maintenance() { let candidate = CseCandidate { - subtree: &scan(), + sub_dag: &scan(), bound_summary: &summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, @@ -1491,7 +1491,7 @@ mod tests { #[test] fn cse_share_decision_recomputes_when_maintenance_dominates_recompute() { let candidate = CseCandidate { - subtree: &scan(), + sub_dag: &scan(), bound_summary: &summary_node(SummaryFamilyType::StatModel( asap_types::post_asap::StatModelKind::Parametric, asap_types::post_asap::StatModelParams::Parametric { @@ -1531,7 +1531,7 @@ mod tests { // actually calls through to the overridable hooks rather than // hardcoding a comparison against its own defaults. let candidate = CseCandidate { - subtree: &scan(), + sub_dag: &scan(), bound_summary: &summary_node(SummaryFamilyType::StatModel( asap_types::post_asap::StatModelKind::Parametric, asap_types::post_asap::StatModelParams::Parametric { @@ -1626,7 +1626,7 @@ mod tests { } /// `DefaultCostModel::estimate_cost` for a [`Replacement::Rewrite`] pair - /// (the `SharedSubtreeStrategy` share-vs-recompute shape) agrees with + /// (the `SharedSubDAGStrategy` share-vs-recompute shape) agrees with /// what `cse_share_decision` would already pick for the same target: with /// many consumers of a cheap-to-recompute leaf, the "share" candidate /// (the target's own `Rc`) must cost less than the "recompute diff --git a/crates/asap-aware-mapping/src/exact_composition.rs b/crates/asap-aware-mapping/src/exact_composition.rs index a4508dc3b..0a683550c 100644 --- a/crates/asap-aware-mapping/src/exact_composition.rs +++ b/crates/asap-aware-mapping/src/exact_composition.rs @@ -412,7 +412,7 @@ impl<'a> ExactCompositionStrategy<'a> { rationale: format!( "{} is an exact fold whose input is the readout of {} — a maintained \ accumulator cannot consume query-time values, so instead of collapsing \ - the whole tree into KeepPreAsap this applies the fold as an \ + the whole DAG into KeepPreAsap this applies the fold as an \ ExactRead over whichever summary readout global_selection \ commits for the child target (asap_aware_mapping::exact_composition)", describe_intent(&intent), diff --git a/crates/asap-aware-mapping/src/explanation.rs b/crates/asap-aware-mapping/src/explanation.rs index 5f930dbc4..633e30d34 100644 --- a/crates/asap-aware-mapping/src/explanation.rs +++ b/crates/asap-aware-mapping/src/explanation.rs @@ -17,7 +17,7 @@ //! //! Earlier (PR #247, superseded by this module — see "What this replaces" //! below), "is optimization X applicable here?" was a yes/no fact each rule -//! re-derived by walking the tree itself. That made sense before there was +//! re-derived by walking the DAG itself. That made sense before there was //! any other structure to consult. But [`crate::replacement::search_workload`] //! (issue #252) now *already* computes, for every //! [`TargetSubDAG`](crate::replacement::TargetSubDAG) in the workload, every @@ -36,7 +36,7 @@ //! candidate isn't an *opportunity*, it's just the target's existing shape //! reflected back. A `TargetSubDAG` with more than one candidate (several //! sketch families to choose between), or one candidate that is itself a -//! genuine alternative to the status quo (share this already-shared subtree +//! genuine alternative to the status quo (share this already-shared sub-DAG //! instead of recomputing it at every consumer), *is* an applicability //! finding — [`explain_replacements`] and //! [`explain_replacements_with`] just translate [`CandidateLogicalASAPDAGs`]'s @@ -50,8 +50,8 @@ //! `realizations_for_intent` would have committed to on its own. //! - [`ExplanationKind::CommonSubexpressionReuse`] — the `TargetSubDAG` //! has two or more consumers *and* its candidate list contains the -//! [`SharedSubtreeStrategy`] "build once and share" candidate (the one -//! whose `Rc` is the group's own `target`) — i.e. sharing this subtree +//! [`SharedSubDAGStrategy`] "build once and share" candidate (the one +//! whose `Rc` is the group's own `target`) — i.e. sharing this sub-DAG //! instead of recomputing it independently is a real, reported choice, not //! just an accident of how the workload happened to be built. //! @@ -76,7 +76,7 @@ //! [`crate::replacement::search_workload`]/[`crate::replacement::search_workload_with`]'s //! own signature shape. Only the *data source* changed: this module now //! calls those two functions and translates the result, rather than running -//! its own rules and their supporting traversal over the tree a second time. +//! its own rules and their supporting traversal over the DAG a second time. //! All of that old traversal is deleted, not kept alongside the new //! implementation — see "Two guarantees the old traversal made, re-verified" //! below for the two properties it's important that deletion didn't quietly @@ -178,7 +178,7 @@ //! [`Replacement::Summary`]: crate::replacement::Replacement::Summary //! [`Replacement::Rewrite`]: crate::replacement::Replacement::Rewrite //! [`SketchAlgorithmStrategy`]: crate::replacement::SketchAlgorithmStrategy -//! [`SharedSubtreeStrategy`]: crate::replacement::SharedSubtreeStrategy +//! [`SharedSubDAGStrategy`]: crate::replacement::SharedSubDAGStrategy //! [`CandidateLogicalASAPDAGs`]: crate::replacement::CandidateLogicalASAPDAGs //! [`TargetSubDAGCandidates`]: crate::replacement::TargetSubDAGCandidates @@ -213,7 +213,7 @@ pub enum ExplanationKind { /// have committed to on its own. SketchApproximation, /// A `TargetSubDAG` has two or more consumers *and* its candidate list - /// contains [`crate::replacement::SharedSubtreeStrategy`]'s "build once + /// contains [`crate::replacement::SharedSubDAGStrategy`]'s "build once /// and share" candidate — the catalog's cross-statistic / cross-metrics / /// cross-subpopulation reuse entries, all the same underlying structural /// fact. @@ -222,7 +222,7 @@ pub enum ExplanationKind { /// [`Replacement::ExactComposition`] — /// [`crate::exact_composition::ExactCompositionStrategy`] found an exact /// operator that can be composed with a summary plan across an explicit - /// update/readout boundary instead of collapsing the whole tree into + /// update/readout boundary instead of collapsing the whole DAG into /// `KeepPreAsap` (issue #171). ExactComposition, } @@ -234,11 +234,11 @@ pub enum ExplanationKind { /// [`crate::replacement::ReplacementSubDAG::rationale`]). /// /// `node_hash` is [`structural_hash`](asap_types::pre_asap::cse::structural_hash) -/// of the `TargetSubDAG`'s own `target` subtree — the same function, on the -/// same `Rc` shape, that [`asap_types::dag_export::DagNode::hash`] +/// of the `TargetSubDAG`'s own `target` sub-DAG — the same function, on the +/// same `Rc` shape, that [`asap_types::dag_export::DAGNode::hash`] /// is computed with. A downstream consumer that independently exported the /// same `QueryExpr` (e.g. via `asap_types::dag_export::export`) can match -/// this explanation to a `DagNode` by first comparing hashes and then +/// this explanation to a `DAGNode` by first comparing hashes and then /// confirming structural equality with [`ReplacementExplanation::target`]. #[derive(Debug, Clone, PartialEq)] pub struct ReplacementExplanation { @@ -455,7 +455,7 @@ fn visit( /// `node`'s own **relational-skeleton** operator children — the same scope /// `crate::replacement`'s own target-discovery `walk_children` (and -/// `asap_types::pre_asap::cse::share_common_subtrees`'s `rebuild_children`) +/// `asap_types::pre_asap::cse::share_common_sub_dags`'s `rebuild_children`) /// use. Exhaustive over every `QueryExpr` variant: a new variant fails to /// compile here until this match is extended too. fn visit_children( @@ -563,15 +563,15 @@ mod tests { } /// `node_hash` must be the literal `structural_hash` a downstream - /// consumer would compute over the *same* `QueryExpr` subtree via + /// consumer would compute over the *same* `QueryExpr` sub-DAG via /// `asap_types::dag_export::export` — the whole point of carrying it is - /// that two independent exports of the same tree agree, with no + /// that two independent exports of the same DAG agree, with no /// string-matching against `location` required. #[test] - fn node_hash_matches_dag_export_hash_for_the_same_subtree() { + fn node_hash_matches_dag_export_hash_for_the_same_sub_dag() { let q = agg(vec![2], default_quantile(0.99), metric_scan(&["job"])); - let graph = asap_types::dag_export::export(&q); - let expected_hash = graph.nodes[graph.root as usize].hash; + let dag = asap_types::dag_export::export(&q); + let expected_hash = dag.nodes[dag.root as usize].hash; let findings = explain_replacements(vec![("dashboard_p99", q)]); let sketch = findings @@ -581,8 +581,8 @@ mod tests { assert_eq!( Some(sketch.node_hash), expected_hash, - "ReplacementExplanation::node_hash must match dag_export's DagNode::hash \ - for the same QueryExpr subtree" + "ReplacementExplanation::node_hash must match dag_export's DAGNode::hash \ + for the same QueryExpr sub-DAG" ); } @@ -637,7 +637,7 @@ mod tests { /// A sketch-applicable `Aggregate` reachable via two paths that CSE /// collapses onto one `Rc` — the same `median(x) == median(x)` shape - /// `pre_asap::cse`'s own `single_query_shares_its_own_repeated_subtree` + /// `pre_asap::cse`'s own `single_query_shares_its_own_repeated_sub_dag` /// test uses — must be reported once, not once per path: it is exactly /// one [`crate::replacement::TargetSubDAGCandidates`], keyed by `Rc` pointer identity, /// not one per path that reaches it. @@ -670,7 +670,7 @@ mod tests { #[test] fn two_roots_with_the_same_grouped_aggregate_share_a_reuse_finding() { // Grouped (`by (job)`), so the shared `Aggregate`'s output schema - // carries a provable unique key — share_common_subtrees's legality + // carries a provable unique key — share_common_sub_dags's legality // gate — and identical across both roots, so it is shareable. let a = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let b = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); @@ -722,7 +722,7 @@ mod tests { #[test] fn ungrouped_identical_aggregates_are_not_shareable_so_no_finding() { - // Empty `by`: no provable unique key — share_common_subtrees never + // Empty `by`: no provable unique key — share_common_sub_dags never // hoists these, so consumer_count stays 1 for each and this module // must not report a finding either. let a = agg(vec![], AggIntent::Sum { col: None }, metric_scan(&["job"])); @@ -762,13 +762,13 @@ mod tests { /// A shared node nested three levels under two *different*, unshared /// `Filter` parents (mirrors `crate::replacement::tests:: - /// nested_shared_subtree_below_an_unshared_parent_is_still_discovered`) + /// nested_shared_sub_dag_below_an_unshared_parent_is_still_discovered`) /// must still be exactly one finding — the maximal-`TargetSubDAG` /// guarantee the module docs describe, now provided by /// `crate::replacement`'s own target discovery rather than this module's /// (deleted) traversal. #[test] - fn a_deeply_shared_subtree_under_different_parents_is_reported_once() { + fn a_deeply_shared_sub_dag_under_different_parents_is_reported_once() { use asap_types::pre_asap::expr_ir::ScalarValue; use asap_types::pre_asap::query_expr::Predicate; diff --git a/crates/asap-aware-mapping/src/grouping.rs b/crates/asap-aware-mapping/src/grouping.rs index b2899f0d4..f74205205 100644 --- a/crates/asap-aware-mapping/src/grouping.rs +++ b/crates/asap-aware-mapping/src/grouping.rs @@ -43,7 +43,7 @@ //! whole-recursive-bind decision procedure toward a specific `SketchKind`, //! the same pattern [`crate::replacement::SketchAlgorithmStrategy`]'s own module //! docs explain was deliberately deleted from this crate as an anti-pattern: -//! forcing a choice via a whole-tree `CostModel` adapter had a real bug where +//! forcing a choice via a whole-DAG `CostModel` adapter had a real bug where //! the forced choice could leak into a target's own nested aggregates. This //! module never needs that: [`crate::replacement::realizations_for_intent`] //! already returns every ranked candidate `Realization` directly, so diff --git a/crates/asap-aware-mapping/src/lib.rs b/crates/asap-aware-mapping/src/lib.rs index 3af72bf5f..95e50a552 100644 --- a/crates/asap-aware-mapping/src/lib.rs +++ b/crates/asap-aware-mapping/src/lib.rs @@ -2,13 +2,13 @@ //! //! This crate sits between the language-agnostic IR ([`asap_ir`]) and //! any runtime: it consumes pre-ASAP [`QueryExpr`](asap_types::pre_asap::QueryExpr) -//! trees and makes the cost-aware decisions the pre-ASAP IR deliberately +//! DAGs and makes the cost-aware decisions the pre-ASAP IR deliberately //! leaves open — which sketch (if any) realises each approximate intent. //! //! **Common sub-expression elimination (CSE) is not this crate's job.** //! Detection is a primary pass over the pre-ASAP `QueryExpr` IR itself //! (`asap_types::pre_asap`, design tracked in issue #223), run before a -//! tree ever reaches [`replacement::SketchAlgorithmStrategy`] — see issue #222 +//! DAG ever reaches [`replacement::SketchAlgorithmStrategy`] — see issue #222 //! for why (batch query optimization needs to see shared work across a //! `QueryWorkload` before summary binding, not after). This crate may //! eventually run a second, narrower CSE pass of its own over an @@ -74,7 +74,7 @@ //! prose), meant for the same downstream consumer (e.g. a //! DAG-visualization view) the crate doc's planning workflows section above //! already names for [`replacement::CandidateLogicalASAPDAGs`] itself. Superseded PR -//! #247's own rule-based traversal, which re-walked the tree once per +//! #247's own rule-based traversal, which re-walked the DAG once per //! optimization before [`replacement::search_workload`] existed to read //! from instead — see that module's docs for the full reframing. //! - [`rollup`] — [`rollup::RollupStrategy`] wraps group-by-lattice roll-up @@ -83,7 +83,7 @@ //! re-deriving it from an already-computed, strictly finer sibling //! `Aggregate` over identical child IR instead of an independent pass //! over the raw source — the cross-aggregate sibling of -//! `pre_asap::cse::share_common_subtrees`'s identical-subtree sharing. +//! `pre_asap::cse::share_common_sub_dags`'s identical-sub-DAG sharing. //! [`rollup::is_legal_rollup_source`] is the standalone legality predicate //! other axes (e.g. issue #256's `GroupingStrategy`) are expected to //! consult directly, so it and this module's `RollupStrategy` can never @@ -101,7 +101,7 @@ //! is a [`replacement::ReplacementStrategy`] that reshapes a bare `avg` //! node — which [`replacement::realizations_for_intent`] can only //! dispatch to `Realization::PassThrough`, so it can never be a -//! [`replacement::SharedSubtreeStrategy`] target — into a `sum`/`count` +//! [`replacement::SharedSubDAGStrategy`] target — into a `sum`/`count` //! pair under the same grouping, re-divided back by a wrapping `Project`, //! so those *are* ordinary mergeable accumulators sharing/sketching can //! reach. It only reshapes; [`replacement::search_workload`]'s cost-based @@ -211,13 +211,13 @@ pub use replacement::{ search_workload_with_targets, summary_candidates, CandidateLogicalASAPDAGs, CompositionDecision, GlobalSelection, Matcher, Proposals, RankedTargetSubDAGCandidates, Realization, RealizationError, RecurrenceProfileMap, RejectedCandidate, Replacement, - ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, SharedSubtreeStrategy, + ReplacementProvenance, ReplacementStrategy, ReplacementSubDAG, SharedSubDAGStrategy, SketchAlgorithmStrategy, TargetSubDAG, TargetSubDAGCandidates, TargetSubDAGSelection, MAX_SEARCH_ITERATIONS, }; pub use rewrite::{AvgToSumOverCountStrategy, SemanticEquivalentRewriteStrategy}; pub use summary_maintenance_dag_export::{ - export_summary_maintenance_plan, SummaryMaintenanceDagExport, + export_summary_maintenance_plan, SummaryMaintenanceDAGExport, SummaryMaintenanceDeploymentExport, SummaryMaintenanceLifecycleAlternativeExport, }; pub use summary_maintenance_lifecycle::{ diff --git a/crates/asap-aware-mapping/src/maintained_population.rs b/crates/asap-aware-mapping/src/maintained_population.rs index 8d9460c14..61c88b0ad 100644 --- a/crates/asap-aware-mapping/src/maintained_population.rs +++ b/crates/asap-aware-mapping/src/maintained_population.rs @@ -312,7 +312,7 @@ impl ReplacementStrategy for MaintainedPopulationStrategy { mod tests { use super::*; use crate::test_support::lower_promql; - use asap_types::post_asap::{compile_post_asap_dag, share_common_summary_subtrees}; + use asap_types::post_asap::{compile_post_asap_dag, share_common_summary_sub_dags}; fn lower(q: &str) -> Rc { Rc::new(lower_promql(q, asap_types::types::AccuracyTarget::Exact)) @@ -364,7 +364,7 @@ mod tests { .target_subdag_candidates() .flat_map(|g| &g.candidates) .any(|c| c.strategy == "MaintainedPopulationStrategy")); - let plans = share_common_summary_subtrees( + let plans = share_common_summary_sub_dags( roots .iter() .enumerate() diff --git a/crates/asap-aware-mapping/src/pass/major.rs b/crates/asap-aware-mapping/src/pass/major.rs index d74091af7..3671ece58 100644 --- a/crates/asap-aware-mapping/src/pass/major.rs +++ b/crates/asap-aware-mapping/src/pass/major.rs @@ -8,7 +8,7 @@ use std::rc::Rc; -use asap_types::post_asap::{share_common_summary_subtrees, SummaryNode}; +use asap_types::post_asap::{share_common_summary_sub_dags, SummaryNode}; use asap_types::pre_asap::query_expr::QueryExpr; use asap_types::types::AccuracyTarget; @@ -95,7 +95,7 @@ impl OptimizationPass for MajorPass { assembled.push(dag); } let interned = - share_common_summary_subtrees(assembled.iter().cloned().enumerate().collect()); + share_common_summary_sub_dags(assembled.iter().cloned().enumerate().collect()); let states: Vec<_> = interned .iter() .map(|(_, dag)| summary_states(dag)) diff --git a/crates/asap-aware-mapping/src/physical_handoff_cost.rs b/crates/asap-aware-mapping/src/physical_handoff_cost.rs index edf4e30a4..207d812f1 100644 --- a/crates/asap-aware-mapping/src/physical_handoff_cost.rs +++ b/crates/asap-aware-mapping/src/physical_handoff_cost.rs @@ -1,8 +1,8 @@ //! Byte estimates at deployment-declared physical handoffs. use crate::analytical_cost::{ - estimate_physical_dag, AnalyticalCostError, EvidenceBackedPhysicalDag, ExecutionMultiplicity, - PhysicalDagNode, + estimate_physical_dag, AnalyticalCostError, EvidenceBackedPhysicalDAG, ExecutionMultiplicity, + PhysicalDAGNode, }; use crate::physical_operator_statistics::{ComparisonScope, OperatorStatistics}; pub use asap_types::resources::{PhysicalHandoffBytes, PhysicalHandoffKind}; @@ -30,7 +30,7 @@ pub struct PhysicalHandoff { #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(deny_unknown_fields)] pub struct PhysicalHandoffNodeEvidence { - pub node: PhysicalDagNode, + pub node: PhysicalDAGNode, pub statistics: OperatorStatistics, #[serde(rename = "boundaries")] pub handoffs: Vec, @@ -99,11 +99,11 @@ pub struct PhysicalHandoffEstimate { } fn invalid(reason: &'static str) -> AnalyticalCostError { - AnalyticalCostError::InvalidPhysicalDag(reason) + AnalyticalCostError::InvalidPhysicalDAG(reason) } pub fn estimate_physical_handoffs( - dag: &EvidenceBackedPhysicalDag, + dag: &EvidenceBackedPhysicalDAG, scope: &ComparisonScope, profile: &PhysicalHandoffProfile, evidence_version: &str, diff --git a/crates/asap-aware-mapping/src/physical_plan_cost_model.rs b/crates/asap-aware-mapping/src/physical_plan_cost_model.rs index ffbe08fc4..cd6bf7cb6 100644 --- a/crates/asap-aware-mapping/src/physical_plan_cost_model.rs +++ b/crates/asap-aware-mapping/src/physical_plan_cost_model.rs @@ -8,8 +8,8 @@ use asap_types::resources::CacheProfile; use crate::analytical_cost::{ estimate_physical_dag_comparison, AnalyticalCostError, - EvidenceBackedPhysicalDag as PhysicalDag, PhysicalDagComparisonEstimate, - PhysicalDagEstimateRequest, PhysicalNodeEvidence, ResourceCalibration, + EvidenceBackedPhysicalDAG as PhysicalDAG, PhysicalDAGComparisonEstimate, + PhysicalDAGEstimateRequest, PhysicalNodeEvidence, ResourceCalibration, }; use crate::cost_model::{Cost, CostModel, DefaultCostModel}; use crate::physical_operator_statistics::ComparisonScope; @@ -59,13 +59,13 @@ pub trait PlannerPhysicalPlanProvider { snapshot: &PhysicalEvidenceSnapshot, summary: &Rc, target: &TargetSubDAG<'_>, - ) -> Result; + ) -> Result; } /// Dimensional comparison retained for explanations and verification. #[derive(Debug, Clone, PartialEq)] pub struct PhysicalPlanComparison { - pub resources: PhysicalDagComparisonEstimate, + pub resources: PhysicalDAGComparisonEstimate, pub raw_cost: Cost, pub candidate_cost: Cost, pub storage_io: Option<( @@ -94,7 +94,7 @@ struct CachedTargetEvidence { root: Rc, consumer_count: usize, snapshot: PhysicalEvidenceSnapshot, - raw: PhysicalDag, + raw: PhysicalDAG, } impl<'a> PhysicalPlanCostModel<'a> { @@ -124,7 +124,7 @@ impl<'a> PhysicalPlanCostModel<'a> { fn target_evidence( &self, target: &TargetSubDAG<'_>, - ) -> Result<(PhysicalEvidenceSnapshot, PhysicalDag), AnalyticalCostError> { + ) -> Result<(PhysicalEvidenceSnapshot, PhysicalDAG), AnalyticalCostError> { if let Some(cached) = self.target_evidence.borrow().iter().find(|cached| { Rc::ptr_eq(&cached.root, target.root) && cached.consumer_count == target.consumer_count }) { @@ -199,14 +199,14 @@ impl<'a> PhysicalPlanCostModel<'a> { }, }; let resources = estimate_physical_dag_comparison( - PhysicalDagEstimateRequest { + PhysicalDAGEstimateRequest { nodes: &raw.nodes, root: &raw.root, scope, statistics: &raw, cache_profile: &snapshot.cache_profile, }, - PhysicalDagEstimateRequest { + PhysicalDAGEstimateRequest { nodes: &replacement.nodes, root: &replacement.root, scope, @@ -365,7 +365,7 @@ mod tests { DataArrival, DurationMs, QueryRecurrence, QueryTimeScope, TimeSelection, TimestampMs, }; - use crate::analytical_cost::{ExecutionMultiplicity, PhysicalDagNode, PhysicalOperator}; + use crate::analytical_cost::{ExecutionMultiplicity, PhysicalDAGNode, PhysicalOperator}; use crate::physical_operator_statistics::{ EdgeStatistics, OperatorStatistics, SourceCoverage, UnaryEdgeStatistics, }; @@ -470,7 +470,7 @@ mod tests { } } - fn summary_dag(&self, scope: &ComparisonScope) -> PhysicalDag { + fn summary_dag(&self, scope: &ComparisonScope) -> PhysicalDAG { let scan_statistics = scan_statistics(self.candidate_scan_bytes, edge(100, 800)); let aggregate_statistics = aggregate_statistics(edge(100, 800), edge(1, 8)); let read_statistics = pass_through_statistics(edge(1, 8)); @@ -500,9 +500,9 @@ mod tests { }, ), ]); - PhysicalDag { + PhysicalDAG { nodes: vec![ - PhysicalDagNode { + PhysicalDAGNode { id: "candidate-scan".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -511,7 +511,7 @@ mod tests { retained_bytes: 0, execution: ExecutionMultiplicity::Once, }, - PhysicalDagNode { + PhysicalDAGNode { id: "candidate-state".into(), operator: PhysicalOperator::HashAggregate { grouping_key_count: 0, @@ -523,7 +523,7 @@ mod tests { retained_bytes: 8, execution: ExecutionMultiplicity::Once, }, - PhysicalDagNode { + PhysicalDAGNode { id: "candidate-read".into(), operator: PhysicalOperator::PassThrough, children: vec!["candidate-state".into()], @@ -585,7 +585,7 @@ mod tests { snapshot: &PhysicalEvidenceSnapshot, _summary: &Rc, _target: &TargetSubDAG<'_>, - ) -> Result { + ) -> Result { assert_eq!(snapshot.version, "test-snapshot-1"); self.summary_available .then(|| self.summary_dag(&snapshot.scope)) @@ -915,7 +915,7 @@ mod tests { snapshot: &PhysicalEvidenceSnapshot, summary: &Rc, target: &TargetSubDAG<'_>, - ) -> Result { + ) -> Result { self.0.summary_physical_dag(snapshot, summary, target) } } @@ -1003,7 +1003,7 @@ mod tests { snapshot: &PhysicalEvidenceSnapshot, summary: &Rc, target: &TargetSubDAG<'_>, - ) -> Result { + ) -> Result { let mut dag = self.0.summary_physical_dag(snapshot, summary, target)?; dag.nodes[0] .source_coverage @@ -1055,7 +1055,7 @@ mod tests { _snapshot: &PhysicalEvidenceSnapshot, _summary: &Rc, _target: &TargetSubDAG<'_>, - ) -> Result { + ) -> Result { panic!("blank snapshot versions must fail before summary binding") } } diff --git a/crates/asap-aware-mapping/src/query_physical_lowering.rs b/crates/asap-aware-mapping/src/query_physical_lowering.rs index aa934e8c9..0aa526059 100644 --- a/crates/asap-aware-mapping/src/query_physical_lowering.rs +++ b/crates/asap-aware-mapping/src/query_physical_lowering.rs @@ -3,8 +3,8 @@ use std::rc::Rc; use crate::analytical_cost::{ - validate_operator_semantics, AnalyticalCostError, EvidenceBackedPhysicalDag, - ExecutionMultiplicity, HashJoinBuildSide, PhysicalDagNode, PhysicalNodeEvidence, + validate_operator_semantics, AnalyticalCostError, EvidenceBackedPhysicalDAG, + ExecutionMultiplicity, HashJoinBuildSide, PhysicalDAGNode, PhysicalNodeEvidence, PhysicalOperator, PromqlBinaryOperandMode, PromqlBinaryOperation, PromqlPresenceKind, PromqlSeriesSampleKind, PromqlVectorCardinality, }; @@ -49,7 +49,7 @@ pub fn lower_query_physical_dag( root: &Rc, scope: &ComparisonScope, evidence: &dyn PhysicalNodeEvidenceProvider, -) -> Result { +) -> Result { use std::collections::HashMap; use asap_types::pre_asap::{GroupKeys, QueryExpr, RelationalSetOpKind}; @@ -61,7 +61,7 @@ pub fn lower_query_physical_dag( provider: &'a dyn PhysicalNodeEvidenceProvider, evidence: HashMap, next_id: usize, - nodes: Vec, + nodes: Vec, } impl Lowerer<'_> { @@ -89,7 +89,7 @@ pub fn lower_query_physical_dag( source_coverage, })?; if evidence.physical_id.is_empty() { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "provider returned an empty physical identity", )); } @@ -104,7 +104,7 @@ pub fn lower_query_physical_dag( source_coverage: Option, ) -> Result { let id = evidence.physical_id.clone(); - let node = PhysicalDagNode { + let node = PhysicalDAGNode { id: id.clone(), operator, children, @@ -115,7 +115,7 @@ pub fn lower_query_physical_dag( }; if let Some(existing) = self.nodes.iter().find(|existing| existing.id == id) { if existing != &node || self.evidence.get(&id) != Some(&evidence) { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "provider reused a physical identity for conflicting evidence", )); } @@ -175,7 +175,7 @@ pub fn lower_query_physical_dag( self.evidence .get(id) .map(|evidence| &evidence.statistics) - .ok_or(AnalyticalCostError::InvalidPhysicalDag( + .ok_or(AnalyticalCostError::InvalidPhysicalDAG( "lowered child statistics are missing", )) } @@ -769,7 +769,7 @@ pub fn lower_query_physical_dag( child_ids: Vec, ) -> Result { if child_ids.is_empty() { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "concat has no children", )); } @@ -823,7 +823,7 @@ pub fn lower_query_physical_dag( }; let root = lowerer.lower(root)?; validate_source_consumption(&lowerer.nodes, scope)?; - Ok(EvidenceBackedPhysicalDag { + Ok(EvidenceBackedPhysicalDAG { nodes: lowerer.nodes, root, evidence: lowerer.evidence, @@ -831,7 +831,7 @@ pub fn lower_query_physical_dag( } fn validate_source_consumption( - nodes: &[PhysicalDagNode], + nodes: &[PhysicalDAGNode], scope: &ComparisonScope, ) -> Result<(), AnalyticalCostError> { let consumed = nodes @@ -841,7 +841,7 @@ fn validate_source_consumption( .collect::>(); for coverage in &consumed { if !scope.sources.contains(coverage) { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "physical scan consumes a source outside the comparison scope", )); } @@ -851,7 +851,7 @@ fn validate_source_consumption( .iter() .any(|expected| !consumed.contains(&expected)) { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "physical scans omit a comparison-scope source", )); } @@ -970,7 +970,7 @@ fn bind_scan_coverage( .cloned() .ok_or_else(|| AnalyticalCostError::ScanOutsideComparisonScope(node_id.into()))?; if matches.any(|candidate| candidate != &coverage) { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "scan source coverage is ambiguous", )); } @@ -1008,7 +1008,7 @@ fn bind_info_coverage( .cloned() .ok_or_else(|| AnalyticalCostError::ScanOutsideComparisonScope(node_id.into()))?; if matches.next().is_some() { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "info source coverage is ambiguous", )); } @@ -1260,7 +1260,7 @@ fn is_promql_scalar(query: &asap_types::pre_asap::QueryExpr) -> bool { mod tests { use super::*; use crate::analytical_cost::{ - estimate_physical_dag, estimate_physical_dag_comparison, PhysicalDagEstimateRequest, + estimate_physical_dag, estimate_physical_dag_comparison, PhysicalDAGEstimateRequest, }; use crate::physical_operator_statistics::{ validate_comparison_scopes, BinaryEdgeStatistics, PartitionStatistics, @@ -1712,14 +1712,14 @@ mod tests { vec!["shared-scan".to_owned(); 2] ); let comparison = estimate_physical_dag_comparison( - PhysicalDagEstimateRequest { + PhysicalDAGEstimateRequest { nodes: &dag.nodes, root: &dag.root, scope: &independent_scope, statistics: &dag, cache_profile: &no_cache, }, - PhysicalDagEstimateRequest { + PhysicalDAGEstimateRequest { nodes: &shared_dag.nodes, root: &shared_dag.root, scope: &shared_scope, @@ -1740,7 +1740,7 @@ mod tests { &shared_scope, &drifted_buffer, ), - Err(AnalyticalCostError::InvalidPhysicalDag( + Err(AnalyticalCostError::InvalidPhysicalDAG( "physical node buffer differs from evidence snapshot" )) ); @@ -1757,7 +1757,7 @@ mod tests { &shared_scope, &drifted_identity, ), - Err(AnalyticalCostError::InvalidPhysicalDag( + Err(AnalyticalCostError::InvalidPhysicalDAG( "evidence map key differs from embedded physical identity" )) ); @@ -1779,7 +1779,7 @@ mod tests { }; assert_eq!( lower_query_physical_dag(&root, &shared_scope, &conflicting_identity), - Err(AnalyticalCostError::InvalidPhysicalDag( + Err(AnalyticalCostError::InvalidPhysicalDAG( "provider reused a physical identity for conflicting evidence" )) ); @@ -2141,7 +2141,7 @@ mod tests { let ambiguous_scope = scope(vec![comparison_scope.sources[0].clone(), second_snapshot]); assert_eq!( lower_query_physical_dag(&root, &ambiguous_scope, &scripted(&conflicting)), - Err(AnalyticalCostError::InvalidPhysicalDag( + Err(AnalyticalCostError::InvalidPhysicalDAG( "scan source coverage is ambiguous" )) ); @@ -2165,7 +2165,7 @@ mod tests { ]); assert_eq!( lower_query_physical_dag(&root, &extra_scope, &scripted(&complete)), - Err(AnalyticalCostError::InvalidPhysicalDag( + Err(AnalyticalCostError::InvalidPhysicalDAG( "physical scans omit a comparison-scope source" )) ); @@ -2619,19 +2619,19 @@ mod tests { assert!(matches!( dag.nodes.as_slice(), [ - PhysicalDagNode { + PhysicalDAGNode { operator: PhysicalOperator::Scan, .. }, - PhysicalDagNode { + PhysicalDAGNode { operator: PhysicalOperator::PromqlRelabel { .. }, .. }, - PhysicalDagNode { + PhysicalDAGNode { operator: PhysicalOperator::PromqlSeriesSample { .. }, .. }, - PhysicalDagNode { + PhysicalDAGNode { operator: PhysicalOperator::PromqlPerSeries { .. }, .. } diff --git a/crates/asap-aware-mapping/src/recurrence.rs b/crates/asap-aware-mapping/src/recurrence.rs index 48e2ba033..56cc67b44 100644 --- a/crates/asap-aware-mapping/src/recurrence.rs +++ b/crates/asap-aware-mapping/src/recurrence.rs @@ -6,7 +6,7 @@ //! neither reached [`CostModel`]'s CSE share-vs-recompute decision //! ([`CostModel::cse_share_decision`]): that decision only ever compared a //! *structural* consumer count (how many workload locations reference a -//! shared subtree) against a flat per-family maintenance weight — it had no +//! shared sub-DAG) against a flat per-family maintenance weight — it had no //! notion of how *often* those consumers actually run. //! //! This module adds that notion as a generic cost context, not a scheduler: @@ -836,13 +836,13 @@ mod tests { #[test] fn decide_falls_back_to_structural_decision_when_profile_is_empty() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 1000, }; @@ -862,13 +862,13 @@ mod tests { #[test] fn decide_rejects_mixed_one_shot_and_repeating_without_horizon() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 2, }; @@ -881,13 +881,13 @@ mod tests { #[test] fn decide_accepts_mixed_one_shot_and_repeating_with_an_explicit_horizon() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 2, }; @@ -942,13 +942,13 @@ mod tests { /// it's read. #[test] fn high_frequency_selects_maintained_low_frequency_selects_recompute() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 1, }; @@ -998,13 +998,13 @@ mod tests { /// cost" acceptance criterion directly against the trait hooks. #[test] fn update_rate_only_affects_maintained_cost_evaluation_rate_affects_both() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 1, }; @@ -1038,13 +1038,13 @@ mod tests { /// `raw_recompute_cost` is expensive. #[test] fn one_shot_only_consumer_decides_without_an_explicit_horizon() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 1, }; @@ -1071,13 +1071,13 @@ mod tests { /// single one-shot consumer strictly prefers `RecomputeIndependently`. #[test] fn one_shot_only_single_consumer_does_not_unconditionally_prefer_share() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 1, }; @@ -1098,13 +1098,13 @@ mod tests { /// own `DeterministicUnitCostModel`. #[test] fn batch_only_workload_does_not_unconditionally_prefer_share_under_default_cost_model() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 1, }; @@ -1175,7 +1175,7 @@ mod tests { /// themselves structurally distinct (so they don't collapse into one /// root the way whole-root-identical fixtures do — see /// `shared_aggregate_across_two_roots_gets_both_strategies_candidates`'s - /// own doc) while letting `share_common_subtrees` unify their + /// own doc) while letting `share_common_sub_dags` unify their /// identical `sum_agg()` children onto one shared `Rc`. fn filtered_root(distinguishing_literal: i64) -> QueryExpr { QueryExpr::Filter { @@ -1366,7 +1366,7 @@ mod tests { assert!(matches!(err, RecurrenceError::InvalidUpdateRate(_))); } - /// Issue #287 review bug 2: a site no root's own structural tree + /// Issue #287 review bug 2: a site no root's own structural DAG /// actually reaches must not have the caller-supplied `update_rate` /// stamped onto it. `AvgToSumOverCountStrategy` (part of /// `default_strategies`, so included by `search_workload`) is a real, @@ -1417,7 +1417,7 @@ mod tests { assert_eq!( count_profile, RecurrenceProfile::EMPTY, - "a site unreachable from any root's own structural tree must fall back to \ + "a site unreachable from any root's own structural DAG must fall back to \ RecurrenceProfile::EMPTY (no update_rate, no evaluation_rate, no one-shot \ consumers), not just an evaluation-rate-free profile that still carries the \ caller's update_rate" @@ -1489,13 +1489,13 @@ mod tests { #[test] fn decide_rejects_a_zero_or_negative_horizon() { - let subtree = scan(); + let sub_dag = scan(); let bound = summary_node(SummaryFamilyType::ExactAggregate( ExactKind::Sum, ExactParams::Sum, )); let candidate = CseCandidate { - subtree: &subtree, + sub_dag: &sub_dag, bound_summary: &bound, consumer_count: 2, }; diff --git a/crates/asap-aware-mapping/src/replacement.rs b/crates/asap-aware-mapping/src/replacement.rs index 809955996..aecf65f89 100644 --- a/crates/asap-aware-mapping/src/replacement.rs +++ b/crates/asap-aware-mapping/src/replacement.rs @@ -34,7 +34,7 @@ //! - [`TargetSubDAG`] — a reference to a pre-ASAP [`QueryExpr`] node that is a //! candidate for replacement, plus how many places in the workload already //! reference it (its `consumer_count`) — the one piece of cross-node -//! context [`SharedSubtreeStrategy`] needs that a bare node reference alone +//! context [`SharedSubDAGStrategy`] needs that a bare node reference alone //! doesn't carry. //! - [`ReplacementSubDAG`] — one candidate replacement for a `TargetSubDAG`: //! either a fully bound [`SummaryNode`] or a pre-ASAP [`QueryExpr`] rewrite @@ -76,16 +76,16 @@ //! ranked list directly: for the same bindable-`Aggregate` shape this crate //! binds (single intent, no `HAVING`), every entry becomes its own bound //! candidate. -//! - [`SharedSubtreeStrategy`] wraps -//! `asap_types::pre_asap::cse::share_common_subtrees`'s sharing decision. +//! - [`SharedSubDAGStrategy`] wraps +//! `asap_types::pre_asap::cse::share_common_sub_dags`'s sharing decision. //! Wherever a [`TargetSubDAG`] already has two or more consumers (i.e. -//! `share_common_subtrees` already collapsed two or more workload +//! `share_common_sub_dags` already collapsed two or more workload //! locations onto the same `Rc` — [`discover_targets`] below //! does the identical workload-wide discovery for [`search_workload_with`]; //! this module's own tests reuse the same dedup logic to build realistic //! fixtures), it reports the two-way candidate CSE's own detection pass //! deliberately declines to pick between on its own: build once and share -//! the already-interned subtree, or build it independently at each +//! the already-interned sub-DAG, or build it independently at each //! consumer. [`crate::cost_model::CostModel::cse_share_decision`] is where //! that choice actually gets made *today* (a fixed comparison, not a //! search) — this strategy exposes the same two-way choice as an explicit, @@ -136,15 +136,15 @@ //! ``` //! //! Read literally, this enumerates whole *plans* — full copies of the -//! workload's tree, one per combination of per-target choices. A workload +//! workload's DAG, one per combination of per-target choices. A workload //! with `N` independently-choosable targets would produce up to `2^N` flat -//! plans, each one duplicating every untouched sibling subtree. This module +//! plans, each one duplicating every untouched sibling sub-DAG. This module //! does not do that: //! //! 1. **Per-target candidates, not flat plans.** [`TargetSubDAGCandidates`] //! stores the alternatives for one distinct [`TargetSubDAG`] (identified by //! its own `Rc` pointer identity — the same currency -//! [`asap_types::pre_asap::cse::share_common_subtrees`] already +//! [`asap_types::pre_asap::cse::share_common_sub_dags`] already //! established across the workload) holding every //! [`ReplacementSubDAG`] alternative discovered for it. [`CandidateLogicalASAPDAGs`] is //! a collection of these groups, keyed by `TargetSubDAG` — a candidate @@ -175,12 +175,12 @@ //! line above stands for: every `TargetSubDAG` this pass discovers is one //! iteration of that loop. It walks every workload root's whole DAG (the //! same **relational-skeleton** operator-child scope -//! `asap_types::pre_asap::cse::share_common_subtrees` itself uses — see +//! `asap_types::pre_asap::cse::share_common_sub_dags` itself uses — see //! that module's "Algorithm" section), discovering one `TargetSubDAG` per //! distinct `Rc` and a *real* `consumer_count`: how many operator-child //! positions anywhere in the workload reference that exact `Rc`, not just //! how many of the workload's own top-level roots happen to be it — a -//! `SharedSubtreeStrategy` candidate three levels under an unshared +//! `SharedSubDAGStrategy` candidate three levels under an unshared //! `Filter` is exactly as real a target as a shared whole root, so this //! module's discovery can't stop at the top level. //! @@ -214,7 +214,7 @@ //! not already known, and any found become next round's frontier. Both shipped //! strategies are idempotent in exactly this sense: [`SketchAlgorithmStrategy`] //! produces terminal [`Replacement::Summary`] candidates (no `QueryExpr` -//! children to scan at all), and [`SharedSubtreeStrategy`]'s two +//! children to scan at all), and [`SharedSubDAGStrategy`]'s two //! [`Replacement::Rewrite`] candidates both reuse the target's own //! already-known child `Rc`s verbatim (`Rc::clone`/a shallow top-level //! `.clone()` — see that strategy's own doc). So for both, the frontier is @@ -244,7 +244,7 @@ //! being the whole story; it isn't a contradiction of #237, it's the scope //! change #237 itself named). Concretely, per [`TargetSubDAGCandidates`]: //! -//! - A group whose candidates are the [`SharedSubtreeStrategy`] +//! - A group whose candidates are the [`SharedSubDAGStrategy`] //! share-vs-recompute pair is ranked by calling //! [`CostModel::cse_share_decision`] via this module's own //! [`cse_preference`] — rather than re-deriving a competing comparison. @@ -265,9 +265,9 @@ //! interact — which both shipped strategies' one-round convergence (see //! "Termination" above) makes the common case — but it's the wrong answer //! whenever they do. Concretely: [`CostModel::cse_share_decision`] costs a -//! [`SharedSubtreeStrategy`] group by comparing a `consumer_count`-scaled +//! [`SharedSubDAGStrategy`] group by comparing a `consumer_count`-scaled //! recompute cost against a fixed maintenance cost — but a **nested** -//! `SharedSubtreeStrategy` group's *true* recompute burden isn't its own +//! `SharedSubDAGStrategy` group's *true* recompute burden isn't its own //! raw [`TargetSubDAGCandidates::consumer_count`] (how many operator-child positions //! directly reference it) whenever an ancestor on the path to it is //! *itself* being recomputed independently rather than shared: recomputing @@ -280,12 +280,12 @@ //! [`CandidateLogicalASAPDAGs::global_selection`] is that missing step: a single //! **top-down dynamic-programming pass** over the discovered sites, //! processed in the topological order [`topological_order`] computes over a -//! small [`ReferenceGraph`] built for exactly this purpose (parent before +//! small [`ReferenceDAG`] built for exactly this purpose (parent before //! every child, so a site's `effective_consumer_count` is always computed //! from *already-decided* ancestors). For every site it computes the //! **effective consumer count** — how many times that site actually runs //! once every ancestor's own selected candidate is accounted for — and, for -//! every [`SharedSubtreeStrategy`]-shaped group, re-decides +//! every [`SharedSubDAGStrategy`]-shaped group, re-decides //! [`CostModel::cse_share_decision`] against *that* corrected count instead //! of the group's raw structural one. When that group also contains a //! non-CSE alternative such as a semantic rewrite, the chosen CSE candidate @@ -296,7 +296,7 @@ //! multiplicity to exactly `1` for everything beneath it (one shared //! execution backs every use of it); a group that chooses //! `RecomputeIndependently` — or has no Share/Recompute decision of its own -//! at all, i.e. isn't itself a `SharedSubtreeStrategy` shape — passes its +//! at all, i.e. isn't itself a `SharedSubDAGStrategy` shape — passes its //! *own* effective count straight through to whatever it references, //! transitively composing contributions from every ancestor on the path, //! not just the immediate parent. @@ -330,10 +330,10 @@ //! groups included. //! - This is not an exhaustive search over combinations of choices for a //! provably-global optimum in every case. [`CostModel::cse_share_decision`] -//! is still a *local*, pairwise comparison at each `SharedSubtreeStrategy` +//! is still a *local*, pairwise comparison at each `SharedSubDAGStrategy` //! site (recompute-total vs. one fixed maintenance cost) — this module //! just now feeds it a *correct* input instead of an *incorrect* one. Two -//! sibling `SharedSubtreeStrategy` groups that could trade off against +//! sibling `SharedSubDAGStrategy` groups that could trade off against //! each other under some shared resource budget (memory, say) still //! aren't jointly optimized here — this crate has no //! cardinality/statistics estimation to bound a combinatorial search like @@ -359,7 +359,7 @@ use asap_types::post_asap::{ use asap_types::post_asap::{AccuracyError, CompositionOperator, GuaranteeSource, ResultGuarantee}; use asap_types::pre_asap::agg_intent::{agg_is_mergeable, AggIntent}; use asap_types::pre_asap::column_resolution::resolve_column_ref; -use asap_types::pre_asap::cse::{share_common_subtrees, structural_hash, HashCache}; +use asap_types::pre_asap::cse::{share_common_sub_dags, structural_hash, HashCache}; use asap_types::pre_asap::expr_ir::{ArithmeticOpKind, ColumnRef}; use asap_types::pre_asap::query_expr::any_measure_filtered; use asap_types::pre_asap::query_expr::{ @@ -426,20 +426,20 @@ pub enum RealizationError { /// A pre-ASAP sub-DAG a [`ReplacementStrategy`] knows how to replace. /// -/// `root` is a reference into the workload's own [`QueryExpr`] tree (an +/// `root` is a reference into the workload's own [`QueryExpr`] DAG (an /// `Rc`, the same currency [`search_workload`] and -/// `asap_types::pre_asap::cse::share_common_subtrees` already thread through +/// `asap_types::pre_asap::cse::share_common_sub_dags` already thread through /// this crate's public API — not a bare `&QueryExpr` — so a strategy that /// needs the node's own `Rc` identity, not just its shape, has it available /// without the caller re-deriving it). /// /// `consumer_count` is how many locations across the workload reference this -/// exact `Rc` — 1 for an ordinary single-use node and 2+ for a shared subtree. +/// exact `Rc` — 1 for an ordinary single-use node and 2+ for a shared sub-DAG. /// [`search_workload_with`] computes the workload-wide value during target /// discovery. [`TargetSubDAG::new`] defaults it to `1` for callers invoking a /// strategy against one node in isolation. A strategy that only cares about /// `root`'s shape (for example, [`SketchAlgorithmStrategy`]) can ignore the -/// count; [`SharedSubtreeStrategy`] consults it directly. +/// count; [`SharedSubDAGStrategy`] consults it directly. /// /// `strictest_sibling_accuracy` is the strictest accuracy among workload /// siblings that read the same summary input as `root`, when stricter than @@ -487,7 +487,7 @@ pub enum Replacement { Summary(Rc), /// A pre-ASAP rewrite: still a logical [`QueryExpr`], structurally /// different from the target's own `root` (e.g. sharing vs. not sharing - /// a subtree) but semantically equivalent to it. + /// a sub-DAG) but semantically equivalent to it. Rewrite(Rc), /// An exact operator composed over another target's *own* selected /// decision across an explicit update/readout boundary (issue #171): @@ -614,7 +614,7 @@ pub struct Proposals { /// of this trait or any existing strategy required. /// /// `replacements` is only meaningful when `matches` would return `true` for -/// the same target; both [`SketchAlgorithmStrategy`] and [`SharedSubtreeStrategy`] +/// the same target; both [`SketchAlgorithmStrategy`] and [`SharedSubDAGStrategy`] /// return an empty `Vec` rather than panicking when called on a target they /// don't match, so a caller that skips the `matches` check first still gets a /// safe (merely uninformative) answer instead of a crash. @@ -1573,7 +1573,7 @@ impl<'a> SketchAlgorithmStrategy<'a> { proposals.record( format!( "{rationale}; sized under {} budget split of {target:?} across \ - {} approximate layers (this layer {outer_target:?}, child subtree \ + {} approximate layers (this layer {outer_target:?}, child sub-DAG \ {inner_target:?})", allocation.allocator, shape.approximate_layer_count ), @@ -1933,7 +1933,7 @@ pub(crate) fn realize_child_with( // produce — never happens, that match is exhaustive), or every // candidate was accuracy-illegal — either way the same conservative // fallback `SketchAlgorithmStrategy::matches` uses: keep the - // pre-ASAP subtree, executed exactly. + // pre-ASAP sub-DAG, executed exactly. None => keep_pre_asap(root), } } @@ -2336,7 +2336,7 @@ fn override_accuracy(intent: &AggIntent, target: &AccuracyTarget) -> AggIntent { out } -/// Wrap an unrewritten pre-ASAP subtree, lifting its schema with every column +/// Wrap an unrewritten pre-ASAP sub-DAG, lifting its schema with every column /// `SummaryFamilyType::Plain`. `pub` so a caller can fall back to this /// explicitly — e.g. when `SketchAlgorithmStrategy::replacements()` returns no /// candidate for a target, or a deployment wants to force a node its own @@ -2351,7 +2351,7 @@ fn keep_pre_asap_rc(expr: Rc) -> Result, RealizationE Ok(Rc::new(SummaryNode { expr: SummaryExpr::KeepPreAsap(expr), schema: lift(&schema), - // A kept pre-ASAP subtree is executed exactly by the runtime + // A kept pre-ASAP sub-DAG is executed exactly by the runtime // (`Realization::PassThrough`'s contract) — zero error. guarantee: Some(ResultGuarantee::exact("KeepPreAsap")), })) @@ -2363,7 +2363,7 @@ fn keep_pre_asap_rc(expr: Rc) -> Result, RealizationE /// `HAVING`. A multi-intent node (SQL `SELECT SUM(a), AVG(b)`), or one with a /// `HAVING` predicate (the filter would need the estimate first), stays /// logical. Unsupported logical parents still conservatively become one -/// [`SummaryExpr::KeepPreAsap`] subtree. Composable query-time value +/// [`SummaryExpr::KeepPreAsap`] sub-DAG. Composable query-time value /// operators (`Project`, `Filter`, `Sort`, and `Limit`) are retained during final /// DAG assembly so their independently planned children remain visible. pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { @@ -2400,7 +2400,7 @@ pub fn bindable_intent(node: &QueryExpr) -> Option<&AggIntent> { /// Construct a summary with every model explicit (issue #172). `intent` /// is `expr`'s own [`bindable_intent`], or a copy of it with an allocated /// `AccuracyTarget` substituted (see [`realize_child_with`]). -/// `child_target`, when set, is the end-to-end budget the child subtree is +/// `child_target`, when set, is the end-to-end budget the child sub-DAG is /// re-enumerated under; `allocation` is the provenance note recording the /// split that produced both. `Err(RealizationError::Accuracy)` is the /// fail-closed answer for a composition with no sound rule or one that @@ -3756,15 +3756,15 @@ fn lift(schema: &Schema) -> SummarySchema { } } -// ── SharedSubtreeStrategy ──────────────────────────────────────────────── +// ── SharedSubDAGStrategy ──────────────────────────────────────────────── -/// Wraps `asap_types::pre_asap::cse::share_common_subtrees`'s sharing +/// Wraps `asap_types::pre_asap::cse::share_common_sub_dags`'s sharing /// decision as an explicit candidate pair, wherever a [`TargetSubDAG`] /// already has two or more consumers. /// /// This strategy does not decide sharing itself, nor does it discover which /// nodes are shared — by the time a caller builds a `TargetSubDAG` with -/// `consumer_count >= 2`, `share_common_subtrees` has already made that +/// `consumer_count >= 2`, `share_common_sub_dags` has already made that /// (legality-gated, `PartialEq`-checked) call; [`discover_targets`] below /// discovers real consumer counts across a workload the same way for /// [`search_workload_with`] (this module's own tests reuse the identical @@ -3774,9 +3774,9 @@ fn lift(schema: &Schema) -> SummarySchema { /// `Rc`" as the two-way choice a downstream cost model (today, /// [`CostModel::cse_share_decision`]) picks between: build once and share, or /// build independently at each consumer. -pub struct SharedSubtreeStrategy; +pub struct SharedSubDAGStrategy; -impl ReplacementStrategy for SharedSubtreeStrategy { +impl ReplacementStrategy for SharedSubDAGStrategy { fn matches(&self, target: &TargetSubDAG<'_>) -> bool { target.consumer_count >= 2 } @@ -3788,27 +3788,27 @@ impl ReplacementStrategy for SharedSubtreeStrategy { let count = target.consumer_count; vec![ ReplacementSubDAG { - strategy: "SharedSubtreeStrategy", + strategy: "SharedSubDAGStrategy", // The already-interned `Rc` itself: reusing it verbatim *is* // "build once and share" — no new node to construct. replacement: Replacement::Rewrite(Rc::clone(target.root)), provenance: ReplacementProvenance::CseShare, rationale: format!( - "build once and share: share_common_subtrees already interned this \ - subtree once and reused it across {count} consumers — one build can \ + "build once and share: share_common_sub_dags already interned this \ + sub-DAG once and reused it across {count} consumers — one build can \ answer all of them instead of computing it {count} times" ), }, ReplacementSubDAG { - strategy: "SharedSubtreeStrategy", + strategy: "SharedSubDAGStrategy", // A structurally-identical but freshly-allocated `Rc`: same // value (`PartialEq`), deliberately *not* the same pointer, // representing "undo the sharing and recompute independently". replacement: Replacement::Rewrite(Rc::new((**target.root).clone())), provenance: ReplacementProvenance::CseRecompute, rationale: format!( - "build independently: undo the sharing share_common_subtrees found and \ - recompute this subtree separately at each of its {count} consumers — \ + "build independently: undo the sharing share_common_sub_dags found and \ + recompute this sub-DAG separately at each of its {count} consumers — \ worth it only when independence outweighs the shared-maintenance cost, \ a CostModel's call (e.g. CostModel::cse_share_decision) and not this \ strategy's" @@ -3827,7 +3827,7 @@ impl ReplacementStrategy for SharedSubtreeStrategy { /// A generous, documented backstop against a hypothetically ill-behaved /// future [`ReplacementStrategy`] (see the module docs' "Termination" /// section) — not a bound either shipped strategy could ever approach. -/// [`SketchAlgorithmStrategy`] and [`SharedSubtreeStrategy`] both converge in +/// [`SketchAlgorithmStrategy`] and [`SharedSubDAGStrategy`] both converge in /// exactly 2 passes over a fixed target set, regardless of workload size. pub const MAX_SEARCH_ITERATIONS: usize = 1_000; @@ -3910,7 +3910,7 @@ impl TargetSubDAGCandidates { /// /// Structural (`QueryExpr`) value equality alone is *not* enough here: this /// module's one shipped multi-candidate `Replacement::Rewrite` source, -/// [`SharedSubtreeStrategy`], deliberately returns **two** candidates that +/// [`SharedSubDAGStrategy`], deliberately returns **two** candidates that /// are value-equal to each other (`build once and share` vs. `build /// independently` — see that strategy's own doc) but represent genuinely /// different physical choices, distinguished *only* by whether the @@ -3932,7 +3932,7 @@ impl TargetSubDAGCandidates { /// (currently hypothetical, since neither shipped strategy causes it) /// case of the exact same alternative being proposed twice. A fresh /// [`HashCache`] per call: this is a pairwise check between two candidates -/// for one group, not a bottom-up pass over a whole tree, so there is no +/// for one group, not a bottom-up pass over a whole DAG, so there is no /// wider traversal to amortize the cache across the way `InternTable`'s own /// use of `structural_hash` does. fn is_duplicate_rewrite( @@ -3980,7 +3980,7 @@ fn is_duplicate_summary(_existing: &Rc, _candidate: &Rc` whose /// group holds its alternatives. pub struct CandidateLogicalASAPDAGs { - /// The workload's roots, after the one `share_common_subtrees` pass + /// The workload's roots, after the one `share_common_sub_dags` pass /// [`search_workload_with`] runs up front — the same post-CSE roots /// every `TargetSubDAG` in `groups` was discovered from. pub roots: Vec<(Id, Rc)>, @@ -4063,18 +4063,18 @@ impl CandidateLogicalASAPDAGs { /// The caller supplies a finite expansion budget; exceeding it is an error, /// never a silently truncated inventory presented as exhaustive. #[derive(Debug)] -pub struct CandidateDagInventory { +pub struct CandidateDAGInventory { pub candidates: Vec)>>, pub rejected_assemblies: Vec, } -type CandidateDagChoice<'a> = (Option<&'a ReplacementSubDAG>, Option>); +type CandidateDAGChoice<'a> = (Option<&'a ReplacementSubDAG>, Option>); impl CandidateLogicalASAPDAGs { pub fn enumerate_candidate_dags( &self, expansion_limit: usize, - ) -> Result, RealizationError> { + ) -> Result, RealizationError> { self.enumerate_candidate_roots(&self.roots, expansion_limit) } @@ -4086,7 +4086,7 @@ impl CandidateLogicalASAPDAGs { &self, id: &Id, expansion_limit: usize, - ) -> Result, RealizationError> { + ) -> Result, RealizationError> { let roots = self .roots .iter() @@ -4105,7 +4105,7 @@ impl CandidateLogicalASAPDAGs { &self, roots: &[(Id, Rc)], expansion_limit: usize, - ) -> Result, RealizationError> { + ) -> Result, RealizationError> { let mut reachable = Vec::new(); let mut nodes = HashMap::new(); let mut counts = HashMap::new(); @@ -4133,7 +4133,7 @@ impl CandidateLogicalASAPDAGs { .collect::>(); // Composition plans carry the proofs established during discovery. // No cost ranking is consulted while expanding these choices. - let options: Vec>> = order + let options: Vec>> = order .iter() .map(|ptr| { let group = &self.groups[ptr]; @@ -4168,7 +4168,7 @@ impl CandidateLogicalASAPDAGs { .ok_or(RealizationError::PhysicalRealization( "candidate expansion budget exceeded; no partial inventory returned", ))?; - let mut inventory = CandidateDagInventory { + let mut inventory = CandidateDAGInventory { candidates: Vec::new(), rejected_assemblies: Vec::new(), }; @@ -4214,7 +4214,7 @@ impl CandidateLogicalASAPDAGs { .collect::, _>>(); match roots { Ok(roots) => { - let roots = asap_types::post_asap::share_common_summary_subtrees(roots); + let roots = asap_types::post_asap::share_common_summary_sub_dags(roots); use std::hash::{Hash, Hasher}; let mut hash = std::collections::hash_map::DefaultHasher::new(); let mut pending = roots @@ -4533,7 +4533,7 @@ impl CandidateLogicalASAPDAGs { /// `self.roots[i]` — the same order [`search_workload`]/ /// [`search_workload_with`] were originally called with (post-CSE /// dedup preserves both root count and order — see - /// `asap_types::pre_asap::cse::share_common_subtrees`'s own + /// `asap_types::pre_asap::cse::share_common_sub_dags`'s own /// `.map(...).collect()` body). This keeps `Id` fully opaque (no `Eq`/ /// `Hash`/`Clone` bound needed on it at all — issue #287's "keep /// caller/query identifiers opaque" requirement) at the cost of the @@ -4568,7 +4568,7 @@ impl CandidateLogicalASAPDAGs { /// effective structural execution rate rather than mere reachability. /// /// **Unreachable sites**: [`CandidateLogicalASAPDAGs`] can contain a site no root's own - /// structural tree actually reaches — e.g. one only ever produced by a + /// structural DAG actually reaches — e.g. one only ever produced by a /// [`Replacement::Rewrite`] candidate a [`ReplacementStrategy`] invented /// (this walk only follows [`TargetSubDAGCandidates::target`]'s own structural /// children, the same scope [`discover_targets`] uses for the original @@ -4873,7 +4873,7 @@ fn rank_group<'a>( return ranked; } - // Shape 1: the exact `SharedSubtreeStrategy` share-vs-recompute pair — + // Shape 1: the exact `SharedSubDAGStrategy` share-vs-recompute pair — // rank via `CostModel::cse_share_decision`, the same comparison // the local CSE ranking path already uses. if cse_candidate_pair(group).is_some() { @@ -4974,14 +4974,14 @@ fn rank_group<'a>( } /// For a group whose candidates are all [`Replacement::Rewrite`] (the -/// [`SharedSubtreeStrategy`] shape): does [`CostModel::cse_share_decision`] +/// [`SharedSubDAGStrategy`] shape): does [`CostModel::cse_share_decision`] /// prefer the candidate that shares `group.target`'s own `Rc` (`true`), or /// the one that recomputes independently (`false`)? `None` when there's no /// real comparison to make — fewer than 2 consumers (mirrors -/// [`SharedSubtreeStrategy::matches`]'s own gate), or `group.target` can't +/// [`SharedSubDAGStrategy::matches`]'s own gate), or `group.target` can't /// actually be bound at all (no candidate and no logical fallback — never /// expected in practice for a target that's already part of a legitimate -/// workload tree, but this degrades to "keep discovery order" rather than +/// workload DAG, but this degrades to "keep discovery order" rather than /// panicking). fn cse_preference(group: &TargetSubDAGCandidates, cost_model: &dyn CostModel) -> Option { if group.consumer_count < 2 { @@ -4989,7 +4989,7 @@ fn cse_preference(group: &TargetSubDAGCandidates, cost_model: &dyn CostModel) -> } let bound = realize_one(&group.target, cost_model)?; let candidate = CseCandidate { - subtree: &group.target, + sub_dag: &group.target, bound_summary: &bound, consumer_count: group.consumer_count, }; @@ -5047,7 +5047,7 @@ fn summary_grouping(node: &SummaryNode) -> Option<&GroupingStrategy> { /// One target sub-DAG's selected choice and usage information — the answer /// [`CandidateLogicalASAPDAGs::global_selection`] commits to for one site, after folding in -/// every ancestor [`SharedSubtreeStrategy`] decision on the path from a +/// every ancestor [`SharedSubDAGStrategy`] decision on the path from a /// workload root to this site. See the module docs' "Whole-plan /// (cross-group) selection" section for the full recurrence. /// @@ -5071,7 +5071,7 @@ pub struct TargetSubDAGSelection<'a> { /// ancestor's own selected candidate is accounted for — see /// [`multiplier`]'s doc for the exact recurrence. Equal to /// `consumer_count` unless some ancestor on a path from a root to this - /// site has a [`SharedSubtreeStrategy`] alternative that chose + /// site has a [`SharedSubDAGStrategy`] alternative that chose /// [`ShareDecision::RecomputeIndependently`]. pub effective_consumer_count: usize, /// The candidate chosen for this target, or `None` when no replacement @@ -5262,7 +5262,7 @@ impl<'a> GlobalSelection<'a> { /// when the operator itself has no summary realization. Its child is /// assembled independently, so a selected summary remains visible /// beneath `Project`/`Filter`/`Sort`/`Limit` instead of being swallowed by - /// one opaque `KeepPreAsap` subtree. + /// one opaque `KeepPreAsap` sub-DAG. fn assemble_residual( &self, target: &Rc, @@ -5655,7 +5655,7 @@ impl CandidateLogicalASAPDAGs { /// "Whole-plan (cross-group) selection" section describes: one /// [`TargetSubDAGSelection`] per discovered site, each ranked against an /// `effective_consumer_count` that accounts for every ancestor - /// [`SharedSubtreeStrategy`] decision on the path to it — unlike + /// [`SharedSubDAGStrategy`] decision on the path to it — unlike /// [`Self::cost_sorted`], whose per-group ranking only ever sees a /// group's own raw [`TargetSubDAGCandidates::consumer_count`]. /// Uncertified DDSketch ratios remain in [`CandidateLogicalASAPDAGs`] for downstream @@ -5695,10 +5695,10 @@ impl CandidateLogicalASAPDAGs { horizon: Option, candidate_costs: Option<&CandidateCostOverrides>, ) -> Result, RecurrenceError> { - let graph = reference_graph(self); - let topo = topological_order(&self.order, &graph); + let dag = reference_dag(self); + let topo = topological_order(&self.order, &dag); - let mut effective_uses = graph.external_root_uses.clone(); + let mut effective_uses = dag.external_root_uses.clone(); let mut chosen_share: HashMap<*const QueryExpr, ShareDecision> = HashMap::new(); let mut groups: HashMap<*const QueryExpr, TargetSubDAGSelection<'_>> = HashMap::new(); let mut context = CompositionContext::default(); @@ -5801,7 +5801,7 @@ impl CandidateLogicalASAPDAGs { .then(|| { decide_with_effective_count(group, effective, cost_model).and_then( |decision| { - let candidate = pick_shared_subtree_candidate(group, decision)?; + let candidate = pick_shared_sub_dag_candidate(group, decision)?; chosen_share.insert(*ptr, decision); Some(candidate) }, @@ -5835,7 +5835,7 @@ impl CandidateLogicalASAPDAGs { }; match decision { Some(decision) => { - let cse = pick_shared_subtree_candidate(group, decision); + let cse = pick_shared_sub_dag_candidate(group, decision); let effective_target = TargetSubDAG::with_consumer_count(&group.target, effective); let logical = group @@ -5881,7 +5881,7 @@ impl CandidateLogicalASAPDAGs { } // `realize_child` couldn't produce even a logical fallback — // not expected in practice for a target that's already - // part of a legitimate workload tree (mirrors + // part of a legitimate workload DAG (mirrors // `cse_preference`'s own doc on this same degrade). // Falling back to ordinary local ranking is still a // valid answer, just not a cross-group-aware one; this @@ -6009,7 +6009,7 @@ fn is_automatically_selectable(candidate: &ReplacementSubDAG, cost_model: &dyn C /// chose [`ShareDecision::RecomputeIndependently`] (each of its own uses /// gets its own independent execution, so referencing it costs as much as /// its *own* full multiplicity), or it has no Share/Recompute decision at -/// all (not a [`SharedSubtreeStrategy`] shape — nothing here collapses +/// all (not a [`SharedSubDAGStrategy`] shape — nothing here collapses /// its multiplicity to one, so whatever multiplicity *its* ancestors /// established simply passes through). /// @@ -6082,7 +6082,7 @@ fn decide_with_effective_count( ) -> Option { let bound = realize_child(&group.target, cost_model).ok()?; let candidate = CseCandidate { - subtree: &group.target, + sub_dag: &group.target, bound_summary: &bound, consumer_count: effective_consumer_count, }; @@ -6100,7 +6100,7 @@ fn decide_group_with_recurrence( return Ok(None); }; let candidate = CseCandidate { - subtree: &group.target, + sub_dag: &group.target, bound_summary: &bound, consumer_count: effective_consumer_count, }; @@ -6111,12 +6111,12 @@ fn decide_group_with_recurrence( )) } -/// The [`SharedSubtreeStrategy`] candidate matching `decision`: the one +/// The [`SharedSubDAGStrategy`] candidate matching `decision`: the one /// that shares `group.target`'s own `Rc` for [`ShareDecision::Share`], the /// freshly-allocated one for [`ShareDecision::RecomputeIndependently`] — /// the same `Rc`-identity distinction [`is_duplicate_rewrite`]'s own doc /// explains is the *only* signal this IR carries for that choice. -fn pick_shared_subtree_candidate( +fn pick_shared_sub_dag_candidate( group: &TargetSubDAGCandidates, decision: ShareDecision, ) -> Option<&ReplacementSubDAG> { @@ -6127,7 +6127,7 @@ fn pick_shared_subtree_candidate( }) } -// ── reference graph + topological order ───────────────────────────────── +// ── reference DAG + topological order ───────────────────────────────── /// The parent/child structure [`CandidateLogicalASAPDAGs::global_selection`]'s DP walks — /// built separately from [`discover_targets`]'s own `order`/`nodes`/`counts` @@ -6135,7 +6135,7 @@ fn pick_shared_subtree_candidate( /// breakdown or direction) rather than extending that already-reviewed, /// already-tested pass. Same "small duplicated traversal over reshaping /// proven code" call as [`is_shared_subtree_group`]. -struct ReferenceGraph { +struct ReferenceDAG { /// child ptr -> `(parent ptr, edge count from that one parent)`, for /// every direct operator-child edge in the relational-skeleton scope /// [`walk_children`] itself uses (an edge count above 1 happens when @@ -6147,75 +6147,68 @@ struct ReferenceGraph { /// traversal. children_of: HashMap<*const QueryExpr, Vec<*const QueryExpr>>, /// How many of the workload's own `roots` point directly at each node — - /// a node's "external" use. Nothing inside the tree decides this (it + /// a node's "external" use. Nothing inside the DAG decides this (it /// isn't a reference from another discovered site), so it's never /// subject to any ancestor's Share/Recompute choice — it's the base /// case [`CandidateLogicalASAPDAGs::global_selection`]'s recurrence starts from. external_root_uses: HashMap<*const QueryExpr, usize>, } -/// Build an ordering graph containing every edge that could be selected: +/// Build an ordering DAG containing every edge that could be selected: /// the original target's edges plus every rewrite candidate's edges. An /// accuracy-reconciliation rewrite points at another discovered memo group, /// so it contributes an edge to that group itself; other rewrites contribute /// their relational children as before. The -/// graph is deliberately only used for topological ordering; effective-use +/// DAG is deliberately only used for topological ordering; effective-use /// counts are propagated through the one candidate actually selected. -fn reference_graph(space: &CandidateLogicalASAPDAGs) -> ReferenceGraph { - let mut graph = ReferenceGraph { +fn reference_dag(space: &CandidateLogicalASAPDAGs) -> ReferenceDAG { + let mut dag = ReferenceDAG { parents_of: HashMap::new(), children_of: HashMap::new(), external_root_uses: HashMap::new(), }; for (_, root) in &space.roots { - *graph - .external_root_uses - .entry(Rc::as_ptr(root)) - .or_insert(0) += 1; + *dag.external_root_uses.entry(Rc::as_ptr(root)).or_insert(0) += 1; } for ptr in &space.order { let group = &space.groups[ptr]; - record_possible_edges(*ptr, &group.target, &mut graph); + record_possible_edges(*ptr, &group.target, &mut dag); for candidate in &group.candidates { if let Replacement::Rewrite(rewrite) = &candidate.replacement { if candidate.provenance == ReplacementProvenance::AccuracyReconciliation { - add_edge(*ptr, Rc::as_ptr(rewrite), 1, &mut graph); + add_edge(*ptr, Rc::as_ptr(rewrite), 1, &mut dag); } else { - record_possible_edges(*ptr, rewrite, &mut graph); + record_possible_edges(*ptr, rewrite, &mut dag); } } } } - graph + dag } /// Record one `parent_ptr -> child` edge (both directions — see -/// [`ReferenceGraph`]'s fields), retaining the greatest multiplicity seen +/// [`ReferenceDAG`]'s fields), retaining the greatest multiplicity seen /// when the target and alternative rewrites expose the same edge. fn add_edge( parent_ptr: *const QueryExpr, child_ptr: *const QueryExpr, edge_count: usize, - graph: &mut ReferenceGraph, + dag: &mut ReferenceDAG, ) { - let siblings = graph.parents_of.entry(child_ptr).or_default(); + let siblings = dag.parents_of.entry(child_ptr).or_default(); match siblings.iter_mut().find(|(p, _)| *p == parent_ptr) { Some((_, count)) => *count = (*count).max(edge_count), None => siblings.push((parent_ptr, edge_count)), } - let kids = graph.children_of.entry(parent_ptr).or_default(); + let kids = dag.children_of.entry(parent_ptr).or_default(); if !kids.contains(&child_ptr) { kids.push(child_ptr); } } -fn record_possible_edges( - parent_ptr: *const QueryExpr, - node: &QueryExpr, - graph: &mut ReferenceGraph, -) { +fn record_possible_edges(parent_ptr: *const QueryExpr, node: &QueryExpr, dag: &mut ReferenceDAG) { for (child_ptr, edge_count) in direct_child_counts(node) { - add_edge(parent_ptr, child_ptr, edge_count, graph); + add_edge(parent_ptr, child_ptr, edge_count, dag); } } @@ -6290,16 +6283,16 @@ fn direct_child_counts(node: &QueryExpr) -> Vec<(*const QueryExpr, usize)> { } /// A topological order over `order` (parent before every child) via Kahn's -/// algorithm on `graph`'s reverse adjacency — needed because +/// algorithm on `dag`'s reverse adjacency — needed because /// [`discover_targets`]'s own `order` is only a valid *discovery* order /// (first-seen-first), not a valid topological one: a node reached via two /// different root paths can have a parent that's discovered *after* it (see /// this function's own test for a worked diamond example), which is exactly /// backwards for [`CandidateLogicalASAPDAGs::global_selection`]'s recurrence. -fn topological_order(order: &[*const QueryExpr], graph: &ReferenceGraph) -> Vec<*const QueryExpr> { +fn topological_order(order: &[*const QueryExpr], dag: &ReferenceDAG) -> Vec<*const QueryExpr> { let mut in_degree: HashMap<*const QueryExpr, usize> = HashMap::new(); for ptr in order { - let degree = graph.parents_of.get(ptr).map(Vec::len).unwrap_or(0); + let degree = dag.parents_of.get(ptr).map(Vec::len).unwrap_or(0); in_degree.insert(*ptr, degree); } @@ -6312,7 +6305,7 @@ fn topological_order(order: &[*const QueryExpr], graph: &ReferenceGraph) -> Vec< let mut topo = Vec::with_capacity(order.len()); while let Some(ptr) = queue.pop_front() { topo.push(ptr); - if let Some(children) = graph.children_of.get(&ptr) { + if let Some(children) = dag.children_of.get(&ptr) { for child in children { if let Some(degree) = in_degree.get_mut(child) { *degree -= 1; @@ -6327,9 +6320,9 @@ fn topological_order(order: &[*const QueryExpr], graph: &ReferenceGraph) -> Vec< assert_eq!( topo.len(), order.len(), - "topological_order: the discovered-site reference graph has a cycle — every QueryExpr \ + "topological_order: the discovered-site reference DAG has a cycle — every QueryExpr \ node is built from Rc children, which can't form one, so this indicates a bug in \ - reference_graph rather than a real cyclic workload", + reference_dag rather than a real cyclic workload", ); topo } @@ -6352,10 +6345,10 @@ fn topological_order(order: &[*const QueryExpr], graph: &ReferenceGraph) -> Vec< /// included here (issue #253) even though it's a /// [`Replacement::Rewrite`]-only strategy with no [`CostModel`] of its own to /// plug in — it's context-free (`matches`/`replacements` need nothing beyond -/// the target itself) exactly like [`SharedSubtreeStrategy`], so it belongs +/// the target itself) exactly like [`SharedSubDAGStrategy`], so it belongs /// in this list rather than being derived per-workload the way /// [`RollupStrategy`] is. Rewriting `avg` into `sum`/`count` upfront is what -/// lets [`SketchAlgorithmStrategy`] and [`SharedSubtreeStrategy`] see a +/// lets [`SketchAlgorithmStrategy`] and [`SharedSubDAGStrategy`] see a /// mergeable accumulator to sketch or share at all — see that module's own /// doc comment for why a bare `avg` node otherwise never becomes a /// [`ReplacementStrategy`] target for anything. @@ -6363,7 +6356,7 @@ pub fn default_strategies() -> Vec> { vec![ Box::new(SketchAlgorithmStrategy::default_cost_model()), Box::new(HydraGroupingStrategy::default_cost_model()), - Box::new(SharedSubtreeStrategy), + Box::new(SharedSubDAGStrategy), Box::new(crate::rewrite::AvgToSumOverCountStrategy), Box::new(ExactCompositionStrategy::default_cost_model()), ] @@ -6378,7 +6371,7 @@ pub fn default_strategies_with<'a>( vec![ Box::new(SketchAlgorithmStrategy::new(cost_model)), Box::new(HydraGroupingStrategy::new(cost_model)), - Box::new(SharedSubtreeStrategy), + Box::new(SharedSubDAGStrategy), Box::new(crate::rewrite::SemanticEquivalentRewriteStrategy), Box::new(ExactCompositionStrategy::new(cost_model)), ] @@ -6409,7 +6402,7 @@ pub fn default_strategies_with_evidence<'a>( evidence, ), ), - Box::new(SharedSubtreeStrategy), + Box::new(SharedSubDAGStrategy), Box::new(crate::rewrite::AvgToSumOverCountStrategy), Box::new(ExactCompositionStrategy::new(cost_model)), ] @@ -6435,10 +6428,10 @@ pub fn search_workload(roots: Vec<(Id, Rc)>) -> CandidateLogicalA /// [`RollupStrategy`] is derived and added automatically after CSE for both /// entry points, because only this function owns the post-CSE sibling set. /// -/// Runs [`share_common_subtrees`] once over `roots` first — so every +/// Runs [`share_common_sub_dags`] once over `roots` first — so every /// strategy (and, transitively, every /// [`crate::explanation::ReplacementExplanation`] a caller reads off the -/// result) sees the same already-deduplicated tree — then discovers every +/// result) sees the same already-deduplicated DAG — then discovers every /// `TargetSubDAG` (see [`discover_targets`]) and runs the /// fixpoint loop the module docs describe, capped at /// [`MAX_SEARCH_ITERATIONS`] passes (see the module docs' "Termination" @@ -6651,7 +6644,7 @@ fn strictest_sibling_accuracy( } fn cse_workload(roots: Vec<(Id, Rc)>) -> Vec<(Id, Rc)> { - // `share_common_subtrees` wants owned `QueryExpr`s, not already-`Rc` + // `share_common_sub_dags` wants owned `QueryExpr`s, not already-`Rc` // roots — the same `Rc::try_unwrap`-with-clone-fallback pattern // `asap_types::pre_asap::cse::intern_child` itself uses to recover an // owned node without cloning in the common (uniquely-owned) case. @@ -6662,7 +6655,7 @@ fn cse_workload(roots: Vec<(Id, Rc)>) -> Vec<(Id, Rc)> (id, expr) }) .collect(); - share_common_subtrees(owned_roots) + share_common_sub_dags(owned_roots) } fn search_cse_workload_with<'s, Id>( @@ -6714,7 +6707,7 @@ fn search_cse_workload_with<'s, Id>( "search_workload: fixpoint search did not converge within {MAX_SEARCH_ITERATIONS} \ rounds — a registered ReplacementStrategy's Replacement::Rewrite candidates keep \ exposing new, never-before-seen descendant structure every round. \ - SketchAlgorithmStrategy/SharedSubtreeStrategy never do this (see replacement.rs's \ + SketchAlgorithmStrategy/SharedSubDAGStrategy never do this (see replacement.rs's \ module docs' \"Termination\" section); check any custom strategies passed to \ search_workload_with.", ); @@ -6816,7 +6809,7 @@ fn search_cse_workload_with<'s, Id>( /// Materialize share/recompute alternatives for descendants whose raw edge /// count is one but whose effective count can exceed one when a repeated /// ancestor is recomputed. We only do this when an ordinary repeated group -/// proves that `SharedSubtreeStrategy` is part of this search's strategy set. +/// proves that `SharedSubDAGStrategy` is part of this search's strategy set. fn add_effective_count_cse_candidates( order: &[*const QueryExpr], groups: &mut HashMap<*const QueryExpr, TargetSubDAGCandidates>, @@ -6867,14 +6860,14 @@ fn add_effective_count_cse_candidates( if potentially_repeated.contains(ptr) && cse_candidate_pair(group).is_none() { let target = Rc::clone(&group.target); let site = TargetSubDAG::with_consumer_count(&target, 2); - for mut candidate in SharedSubtreeStrategy.replacements(&site) { + for mut candidate in SharedSubDAGStrategy.replacements(&site) { candidate.rationale = format!( - "{}: this subtree can become repeated when a repeated ancestor is recomputed; \ + "{}: this sub-DAG can become repeated when a repeated ancestor is recomputed; \ global_selection decides using its effective consumer count", match candidate.provenance { ReplacementProvenance::CseShare => "build once and share", ReplacementProvenance::CseRecompute => "recompute independently", - _ => unreachable!("SharedSubtreeStrategy only emits CSE candidates"), + _ => unreachable!("SharedSubDAGStrategy only emits CSE candidates"), } ); group.add_candidate(candidate); @@ -6935,7 +6928,7 @@ fn walk( } /// `node`'s own **relational-skeleton** operator children — the same scope -/// `asap_types::pre_asap::cse::share_common_subtrees`/`rebuild_children` +/// `asap_types::pre_asap::cse::share_common_sub_dags`/`rebuild_children` /// itself uses (see that module's "Algorithm" section) and /// `tests::count_consumers` mirrors for its own fixtures. Exhaustive over /// every `QueryExpr` variant: a new variant fails to compile here until this @@ -7957,7 +7950,7 @@ mod tests { ); } - // ── SketchAlgorithmStrategy / SharedSubtreeStrategy fixtures ─────────── + // ── SketchAlgorithmStrategy / SharedSubDAGStrategy fixtures ─────────── fn metric_scan(labels: &[&str]) -> QueryExpr { let mut columns = vec![ @@ -8272,7 +8265,7 @@ mod tests { } } - // ── SharedSubtreeStrategy ──────────────────────────────────────────── + // ── SharedSubDAGStrategy ──────────────────────────────────────────── #[test] fn does_not_match_a_single_consumer_target() { @@ -8283,8 +8276,8 @@ mod tests { )); let target = TargetSubDAG::new(&q); assert_eq!(target.consumer_count, 1); - assert!(!SharedSubtreeStrategy.matches(&target)); - assert!(SharedSubtreeStrategy.replacements(&target).is_empty()); + assert!(!SharedSubDAGStrategy.matches(&target)); + assert!(SharedSubDAGStrategy.replacements(&target).is_empty()); } #[test] @@ -8295,9 +8288,9 @@ mod tests { metric_scan(&["job"]), )); let target = TargetSubDAG::with_consumer_count(&q, 2); - assert!(SharedSubtreeStrategy.matches(&target)); + assert!(SharedSubDAGStrategy.matches(&target)); - let replacements = SharedSubtreeStrategy.replacements(&target); + let replacements = SharedSubDAGStrategy.replacements(&target); assert_eq!(replacements.len(), 2, "{replacements:?}"); let shared = match &replacements[0].replacement { @@ -8333,7 +8326,7 @@ mod tests { metric_scan(&["job"]), )); let target = TargetSubDAG::with_consumer_count(&q, 3); - let replacements = SharedSubtreeStrategy.replacements(&target); + let replacements = SharedSubDAGStrategy.replacements(&target); assert!(replacements[0].rationale.contains('3')); assert!(replacements[1].rationale.contains('3')); } @@ -8341,7 +8334,7 @@ mod tests { /// Builds realistic multi-consumer `TargetSubDAG`s the same way this /// module's own [`discover_targets`]/`walk` does: dedup by `Rc::as_ptr`, /// walking only the relational-skeleton operator children - /// `asap_types::pre_asap::cse::share_common_subtrees` itself scopes to, + /// `asap_types::pre_asap::cse::share_common_sub_dags` itself scopes to, /// so a shared node nested below another shared node is only ever /// counted at the highest (maximal) point sharing starts. Test-only: /// this module deliberately does not ship a workload-wide discovery @@ -8411,12 +8404,12 @@ mod tests { #[test] fn realistic_cse_output_produces_a_two_consumer_target() { - // Two workload roots that `share_common_subtrees` collapses onto one + // Two workload roots that `share_common_sub_dags` collapses onto one // Rc (mirrors `explanation`'s and `cse`'s own fixtures): a grouped // Sum aggregate over the same scan, built independently at each root. let a = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let b = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); - let shared = asap_types::pre_asap::cse::share_common_subtrees(vec![("a", a), ("b", b)]); + let shared = asap_types::pre_asap::cse::share_common_sub_dags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; @@ -8428,8 +8421,8 @@ mod tests { assert_eq!(count, 2); let target = TargetSubDAG::with_consumer_count(&roots[0], count); - assert!(SharedSubtreeStrategy.matches(&target)); - assert_eq!(SharedSubtreeStrategy.replacements(&target).len(), 2); + assert!(SharedSubDAGStrategy.matches(&target)); + assert_eq!(SharedSubDAGStrategy.replacements(&target).len(), 2); } // ── search_workload / CandidateLogicalASAPDAGs / TargetSubDAGCandidates (merged from search.rs) ── @@ -8586,10 +8579,10 @@ mod tests { #[test] fn shared_aggregate_across_two_roots_gets_both_strategies_candidates() { // Two independently-built, structurally identical Sum aggregates: - // share_common_subtrees (run inside search_workload) collapses them + // share_common_sub_dags (run inside search_workload) collapses them // onto one Rc with consumer_count 2, so this single group should // carry SketchAlgorithmStrategy's one ExactAggregate candidate *and* - // SharedSubtreeStrategy's share-vs-recompute pair. + // SharedSubDAGStrategy's share-vs-recompute pair. let a = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let b = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); @@ -8632,7 +8625,7 @@ mod tests { } #[test] - fn nested_shared_subtree_below_an_unshared_parent_is_still_discovered() { + fn nested_shared_sub_dag_below_an_unshared_parent_is_still_discovered() { // A shared grouped Aggregate nested under two *different*, // unshared Filter parents — real consumer_count must come from // walking the whole DAG, not just root-level pointer identity @@ -8665,7 +8658,7 @@ mod tests { "2 distinct Filters + 1 shared Aggregate + 1 shared Scan" ); - // `share_common_subtrees` re-clones+re-interns anything that already + // `share_common_sub_dags` re-clones+re-interns anything that already // had more than one owner going in (see `cse.rs`'s own doc on // `intern_child`'s clone-fallback path) — so the post-CSE shared // node is a *fresh* Rc, structurally equal to (but not the same @@ -8696,7 +8689,7 @@ mod tests { .expect("shared node must be a discovered target"); assert_eq!(group.consumer_count, 2); assert!( - SharedSubtreeStrategy.matches(&TargetSubDAG::with_consumer_count( + SharedSubDAGStrategy.matches(&TargetSubDAG::with_consumer_count( post_cse_shared, group.consumer_count )) @@ -8707,7 +8700,7 @@ mod tests { #[test] fn add_candidate_rejects_a_true_rewrite_duplicate() { - // SharedSubtreeStrategy's `Replacement::Rewrite` candidates are + // SharedSubDAGStrategy's `Replacement::Rewrite` candidates are // real `QueryExpr` values with `PartialEq`, so `add_candidate` can // (and must) actually reject a genuine repeat — unlike the // `Replacement::Summary` case (see the test below). @@ -8719,7 +8712,7 @@ mod tests { let mut group = TargetSubDAGCandidates::new(Rc::clone(&root), 2); let target = TargetSubDAG::with_consumer_count(&root, 2); let mut inserted = 0; - for candidate in SharedSubtreeStrategy.replacements(&target) { + for candidate in SharedSubDAGStrategy.replacements(&target) { if group.add_candidate(candidate) { inserted += 1; } @@ -8732,7 +8725,7 @@ mod tests { // structurally identical value, both already covered by // `is_duplicate_rewrite`. let mut re_inserted = 0; - for candidate in SharedSubtreeStrategy.replacements(&target) { + for candidate in SharedSubDAGStrategy.replacements(&target) { if group.add_candidate(candidate) { re_inserted += 1; } @@ -8812,7 +8805,7 @@ mod tests { // ── cost-based ranking ─────────────────────────────────────────────── #[test] - fn cost_sorted_orders_shared_subtree_candidates_by_cse_share_decision() { + fn cost_sorted_orders_shared_sub_dag_candidates_by_cse_share_decision() { // Many consumers of a cheap-to-recompute, cheap-to-maintain exact // accumulator: cse_share_decision should prefer Share (see // cost_model.rs's own `cse_share_decision_shares_when_recompute_dominates_maintenance`). @@ -8973,9 +8966,9 @@ mod tests { // ── global_selection (issue #271) ─────────────────────────────────── - /// A `CostModel` with a constant, `subtree`-independent recompute cost + /// A `CostModel` with a constant, `sub_dag`-independent recompute cost /// and shared-maintenance cost, chosen (40 recompute-per-use, 100 - /// maintenance) so that a `SharedSubtreeStrategy` group's + /// maintenance) so that a `SharedSubDAGStrategy` group's /// `cse_share_decision` flips exactly between a consumer count of 2 /// (recompute total 80, below maintenance: `RecomputeIndependently`) /// and a consumer count of 3 (recompute total 120, above @@ -9176,7 +9169,7 @@ mod tests { vec!["recompute", "different rewrite strategy", "share"], "the preferred CSE choice must be ranked without losing the unrelated rewrite" ); - let chosen = pick_shared_subtree_candidate( + let chosen = pick_shared_sub_dag_candidate( &group, decide_with_effective_count(&group, 2, &ConstantCseCost).unwrap(), ) @@ -9186,9 +9179,9 @@ mod tests { #[test] fn effective_consumer_count_corrects_a_nested_groups_share_decision() { - // The interaction issue #271 describes: an outer shared subtree `a` + // The interaction issue #271 describes: an outer shared sub-DAG `a` // (referenced by 2 roots, so consumer_count == 2) wraps an inner - // shared subtree `c` (referenced once through `a`'s own child edge, + // shared sub-DAG `c` (referenced once through `a`'s own child edge, // plus once more directly by a third, separate root — so `c`'s own // *raw* structural consumer_count is also 2, independent of `a`). // @@ -9199,11 +9192,11 @@ mod tests { // // `a` and `c` are both non-`Aggregate` nodes (`Filter`/`Dedup`) so // neither is bindable — each group is a *clean* two-candidate - // SharedSubtreeStrategy share-vs-recompute pair, with no + // SharedSubDAGStrategy share-vs-recompute pair, with no // SketchAlgorithmStrategy `Summary` candidate mixed in to complicate // ranking (see `shared_aggregate_across_two_roots_gets_both_strategies_candidates` // for what a *mixed*-shape group looks like — deliberately avoided - // here to isolate the SharedSubtreeStrategy-only interaction). + // here to isolate the SharedSubDAGStrategy-only interaction). // // Under ConstantCseCost, consumer_count == 2 loses to maintenance // (2 * 40 = 80 < 100 ⇒ RecomputeIndependently); consumer_count == 3 wins @@ -9237,7 +9230,7 @@ mod tests { // Fixture sanity: root1/root2 merged onto one shared `a`, and `c` // (root1/root2's shared child, and root3 itself) merged onto one // shared `c` with raw consumer_count 2, and both groups are clean - // (non-mixed) two-candidate SharedSubtreeStrategy pairs. + // (non-mixed) two-candidate SharedSubDAGStrategy pairs. assert!(Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1)); let a_rc = &space.roots[0].1; let QueryExpr::Filter { child: c_via_a, .. } = a_rc.as_ref() else { @@ -9691,7 +9684,7 @@ mod tests { #[test] fn topological_order_puts_a_later_discovered_parent_before_its_child() { - // Mirrors nested_shared_subtree_below_an_unshared_parent_is_still_discovered's + // Mirrors nested_shared_sub_dag_below_an_unshared_parent_is_still_discovered's // diamond fixture: discover_targets's own `order` visits root_b (a // parent of `shared`) *after* `shared` itself, because `shared` was // already fully walked via root_a first. A naive "process @@ -9731,7 +9724,7 @@ mod tests { order: order.clone(), composition_plans: Vec::new(), }; - let graph = reference_graph(&space); + let dag = reference_dag(&space); // Discovery-order sanity: root_b comes after the shared child in // discover_targets's own order (the exact non-topological case this @@ -9752,7 +9745,7 @@ mod tests { "fixture sanity: discover_targets's own order must NOT already be topological here" ); - let topo = topological_order(&order, &graph); + let topo = topological_order(&order, &dag); let shared_topo_pos = topo.iter().position(|p| *p == shared_ptr).unwrap(); let root_b_topo_pos = topo.iter().position(|p| *p == root_b_ptr).unwrap(); assert!( @@ -10188,7 +10181,7 @@ mod tests { #[test] fn nested_aggregates_realize_per_node() { // quantile(0.9, sum by (job) (m)) — the realization decision - // fires per node over the nested tree: KLL over an exact Sum + // fires per node over the nested DAG: KLL over an exact Sum // accumulator. let inner = agg(vec![2], AggIntent::Sum { col: None }, metric_scan(&["job"])); let outer = agg(vec![], default_quantile(0.9), inner); @@ -10259,7 +10252,7 @@ mod tests { } } - /// The update expression of the first `SummaryAgg` in the tree. + /// The update expression of the first `SummaryAgg` in the DAG. fn find_summary_input(node: &SummaryNode) -> Option { match &node.expr { SummaryExpr::SummaryAgg { input, .. } if input.item.is_none() => { @@ -10274,7 +10267,7 @@ mod tests { fn pass_through_intents_stay_logical() { // avg is exact but non-mergeable; histogram_quantile (classic // buckets, #79) is never sketchable; exact quantile is exact by - // decree. All three stay whole logical subtrees. + // decree. All three stay whole logical sub-DAGs. for intent in [ AggIntent::Avg { col: None }, AggIntent::HistogramQuantile { q: 0.99, le: 0 }, @@ -10296,7 +10289,7 @@ mod tests { #[test] fn logical_parent_subsumes_bindable_child() { // Filter over a bindable quantile: `KeepPreAsap` has no post-ASAP - // children, so the conservative fallback keeps the whole subtree + // children, so the conservative fallback keeps the whole sub-DAG // logical. use asap_types::pre_asap::expr_ir::{CompareOpKind, ScalarValue}; use asap_types::pre_asap::query_expr::Predicate; @@ -10880,7 +10873,7 @@ mod tests { rejection.error ); } - // Fallback keeps the whole subtree pre-ASAP — executed exactly. + // Fallback keeps the whole sub-DAG pre-ASAP — executed exactly. let realized = realize_child(&outer, &DefaultCostModel).unwrap(); assert!(matches!(realized.expr, SummaryExpr::KeepPreAsap(_))); assert!(realized diff --git a/crates/asap-aware-mapping/src/rewrite.rs b/crates/asap-aware-mapping/src/rewrite.rs index 06d1724fd..83c73156b 100644 --- a/crates/asap-aware-mapping/src/rewrite.rs +++ b/crates/asap-aware-mapping/src/rewrite.rs @@ -12,12 +12,12 @@ //! comment on why: `Avg`/`StdDev`/`Variance` "need richer partial state" //! than a bare sketch/exact accumulator gives, so there is no summary //! realization for a bare `avg` node to bind to at all. A logical `avg` -//! node therefore can never be a [`SharedSubtreeStrategy`] target either: +//! node therefore can never be a [`SharedSubDAGStrategy`] target either: //! CSE-style sharing needs *some* mergeable accumulator underneath, and //! `PassThrough` has none. //! //! `Sum` and `Count` are both ordinary mergeable accumulators -//! (`agg_is_mergeable`) — exactly the shape [`SharedSubtreeStrategy`] and a +//! (`agg_is_mergeable`) — exactly the shape [`SharedSubDAGStrategy`] and a //! future sketch-family search already know how to reuse across a //! workload. Rewriting `Aggregate{ measures: [Avg{col}], .. }` into two //! independent single-measure `Sum` and `Count` aggregates, divided with a @@ -43,7 +43,7 @@ //! Both are follow-ups (issue #253 itself scopes to "the concrete case in //! Peilin's comment"), not correctness bugs in what ships here — a node //! outside this scope simply doesn't `match`, the same "safe but -//! uninformative" fallback [`SketchAlgorithmStrategy`]/[`SharedSubtreeStrategy`] +//! uninformative" fallback [`SketchAlgorithmStrategy`]/[`SharedSubDAGStrategy`] //! already use for shapes they don't have an opinion on. //! //! ## Non-goals (mirrors [`replacement`]'s own discipline) @@ -112,7 +112,7 @@ fn avg_rewrite_target(node: &QueryExpr) -> Option<(usize, Option)> { Some((by.keys().len(), *col)) } -/// Build the rewritten `Project{ cast(sum) } / Aggregate{ Count }` tree for +/// Build the rewritten `Project{ cast(sum) } / Aggregate{ Count }` DAG for /// `root`, or `None` if `root` isn't [`avg_rewrite_target`]'s shape. `Sum` and /// `Count` deliberately live in separate, single-measure aggregates so the /// replacement fixpoint discovers each as an independently bindable target. @@ -362,7 +362,7 @@ pub(crate) fn composed_aggregate_rewrite(root: &Rc) -> Option Option< /// 4. `finer_output_schema` (the finer aggregate's own *output* schema, not /// the shared child's) carries a provable unique key /// ([`Schema::has_unique_key`]) — **the exact legality gate -/// `pre_asap::cse::share_common_subtrees` already applies to its own +/// `pre_asap::cse::share_common_sub_dags` already applies to its own /// sharing decisions**, reused verbatim here rather than re-invented: -/// `share_common_subtrees`'s own doc ("Legality: gated by +/// `share_common_sub_dags`'s own doc ("Legality: gated by /// `Schema::unique_keys`") states a producer's output is only safely /// reusable across consumers when its row identity is provably stable — /// exactly the property re-aggregating over `finer` as if it were a /// fresh source requires. /// 5. `coarser_by` is a **strict, proper** subset of `finer_by` (same /// `ColumnId`s, finer strictly more of them) — an *equal* `by` is -/// `SharedSubtreeStrategy`'s CSE-sharing question, not a roll-up, so +/// `SharedSubDAGStrategy`'s CSE-sharing question, not a roll-up, so /// equality is deliberately excluded here, not treated as a degenerate /// roll-up. pub fn is_legal_rollup_source( @@ -511,7 +511,7 @@ mod tests { #[test] fn predicate_rejects_equal_by_sets() { - // Equality is `SharedSubtreeStrategy`'s question, not a roll-up. + // Equality is `SharedSubDAGStrategy`'s question, not a roll-up. let finer_schema = Schema::with_time_index(vec![], 0, vec![vec![0]]); assert!(!is_legal_rollup_source( &GroupKeys::by(vec![2]), @@ -827,7 +827,7 @@ mod tests { #[test] fn equal_by_sets_do_not_roll_up() { - // Equal groupings are `SharedSubtreeStrategy`'s CSE-sharing + // Equal groupings are `SharedSubDAGStrategy`'s CSE-sharing // question (build once and share, or build independently) — a // roll-up requires a *strict* superset, not equality. let scan = Rc::new(metric_scan()); diff --git a/crates/asap-aware-mapping/src/storage_io.rs b/crates/asap-aware-mapping/src/storage_io.rs index 5ff603ca9..646a3532c 100644 --- a/crates/asap-aware-mapping/src/storage_io.rs +++ b/crates/asap-aware-mapping/src/storage_io.rs @@ -8,8 +8,8 @@ use serde::{Deserialize, Serialize}; pub use asap_types::resources::StorageResources; use crate::analytical_cost::{ - estimate_physical_dag, AnalyticalCostError, EvidenceBackedPhysicalDag, ExecutionMultiplicity, - PhysicalDagNode, PhysicalOperator, + estimate_physical_dag, AnalyticalCostError, EvidenceBackedPhysicalDAG, ExecutionMultiplicity, + PhysicalDAGNode, PhysicalOperator, }; use crate::physical_operator_statistics::{ComparisonScope, OperatorStatistics}; @@ -83,7 +83,7 @@ pub struct StorageIoProfile { #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(deny_unknown_fields)] pub struct StorageNodeEvidence { - pub node: PhysicalDagNode, + pub node: PhysicalDAGNode, pub statistics: OperatorStatistics, pub accesses: Vec, } @@ -99,7 +99,7 @@ pub struct StorageEstimate { } fn invalid(reason: &'static str) -> AnalyticalCostError { - AnalyticalCostError::InvalidPhysicalDag(reason) + AnalyticalCostError::InvalidPhysicalDAG(reason) } /// ceil(bytes/request_size), rounded independently per extent and execution. @@ -116,7 +116,7 @@ pub fn request_count(access: &StorageAccess) -> Result } pub fn estimate_storage_io( - dag: &EvidenceBackedPhysicalDag, + dag: &EvidenceBackedPhysicalDAG, scope: &ComparisonScope, profile: &StorageIoProfile, evidence_version: &str, diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs index e31606404..5bf42d2e4 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/estimator.rs @@ -888,7 +888,7 @@ pub(super) fn estimate_transient_liveness( let child_id = summary_physical_id(child, evidence)?; let remaining = uses.get_mut(&child_id) - .ok_or(AnalyticalCostError::InvalidPhysicalDag( + .ok_or(AnalyticalCostError::InvalidPhysicalDAG( "missing summary consumer count", ))?; *remaining -= 1; @@ -1258,7 +1258,7 @@ fn count_operations(root: &SummaryNode) -> Result { if children.is_empty() { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "summary merge has no children", )); } diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs index de20a94a0..f396e3cd8 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/evidence.rs @@ -229,11 +229,11 @@ impl SummaryOperatorEvidence { } } -/// Non-aggregation work for a retained pre-ASAP subtree over the comparison +/// Non-aggregation work for a retained pre-ASAP sub-DAG over the comparison /// horizon. Bootstrap/source I/O belongs exclusively to the owning aggregate, /// and summary insertion belongs exclusively to its insert evidence. #[derive(Debug, Clone, PartialEq)] -pub struct RetainedSubDagEvidence { +pub struct RetainedSubDAGEvidence { pub physical_id: String, /// Logical output edge consumed by the parent summary operator. pub output: EdgeStatistics, @@ -251,7 +251,7 @@ pub struct SummaryNodeEvidence { pub(super) joins: HashMap<*const SummaryNode, SummaryJoinEvidence>, pub(super) operations: HashMap<*const SummaryNode, SummaryOperatorEvidence>, pub(super) operation_state_owners: HashMap<*const SummaryNode, *const SummaryNode>, - pub(super) retained_queries: HashMap<*const SummaryNode, RetainedSubDagEvidence>, + pub(super) retained_queries: HashMap<*const SummaryNode, RetainedSubDAGEvidence>, } impl SummaryNodeEvidence { @@ -287,7 +287,7 @@ impl SummaryNodeEvidence { pub fn insert_retained_query( &mut self, node: &Rc, - evidence: RetainedSubDagEvidence, + evidence: RetainedSubDAGEvidence, ) { self.retained_queries.insert(Rc::as_ptr(node), evidence); } @@ -351,7 +351,7 @@ pub struct RawInputEvidence { /// logical width so compression and encoding are not silently conflated. pub arriving_source_row_bytes: u64, pub ingestion_rate_per_second: f64, - pub physical_dag: EvidenceBackedPhysicalDag, + pub physical_dag: EvidenceBackedPhysicalDAG, } /// One complete provider-enumerated physical implementation of the selected diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs index 6e4901f48..33a7f85c8 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/mod.rs @@ -24,7 +24,7 @@ use crate::analytical_cost::ExecutionMultiplicity; #[cfg(test)] use crate::analytical_cost::PhysicalNodeEvidence; use crate::analytical_cost::{ - estimate_physical_dag, AnalyticalCostError, EvidenceBackedPhysicalDag, PhysicalDagNode, + estimate_physical_dag, AnalyticalCostError, EvidenceBackedPhysicalDAG, PhysicalDAGNode, PhysicalOperator, ResourceCalibration, ResourceEstimate, }; use crate::cost_model::{ diff --git a/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs b/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs index 8bbca6cc4..99917cf20 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_cost/model.rs @@ -155,7 +155,7 @@ fn validate_query_scope( } fn validate_physical_scope_coverage( - physical: &EvidenceBackedPhysicalDag, + physical: &EvidenceBackedPhysicalDAG, scope: &ComparisonScope, ) -> Result<(), AnalyticalCostError> { let nodes = reachable_physical_nodes(physical)?; @@ -184,33 +184,33 @@ fn validate_physical_scope_coverage( } fn reachable_physical_nodes( - physical: &EvidenceBackedPhysicalDag, -) -> Result, AnalyticalCostError> { + physical: &EvidenceBackedPhysicalDAG, +) -> Result, AnalyticalCostError> { let by_id: HashMap<_, _> = physical .nodes .iter() .map(|node| (node.id.as_str(), node)) .collect(); if by_id.len() != physical.nodes.len() { - return Err(AnalyticalCostError::InvalidPhysicalDag("duplicate node id")); + return Err(AnalyticalCostError::InvalidPhysicalDAG("duplicate node id")); } fn visit<'a>( id: &'a str, - by_id: &HashMap<&'a str, &'a PhysicalDagNode>, + by_id: &HashMap<&'a str, &'a PhysicalDAGNode>, visiting: &mut HashSet<&'a str>, visited: &mut HashSet<&'a str>, - nodes: &mut Vec<&'a PhysicalDagNode>, + nodes: &mut Vec<&'a PhysicalDAGNode>, ) -> Result<(), AnalyticalCostError> { if visited.contains(id) { return Ok(()); } if !visiting.insert(id) { - return Err(AnalyticalCostError::InvalidPhysicalDag("cycle")); + return Err(AnalyticalCostError::InvalidPhysicalDAG("cycle")); } let node = by_id .get(id) .copied() - .ok_or(AnalyticalCostError::InvalidPhysicalDag("missing node"))?; + .ok_or(AnalyticalCostError::InvalidPhysicalDAG("missing node"))?; for child in &node.children { visit(child, by_id, visiting, visited, nodes)?; } @@ -291,7 +291,7 @@ fn validate_raw_snapshot_dimensions( .iter() .any(|node| node.execution != ExecutionMultiplicity::Once) { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "streaming raw horizon evidence must use once-counted aggregate statistics", )); } @@ -1607,7 +1607,7 @@ mod tests { first_scan, second_scan, unreachable, - PhysicalDagNode { + PhysicalDAGNode { id: "raw-concat".into(), operator: PhysicalOperator::Concat, children: vec!["raw-scan".into(), "raw-scan-2".into()], @@ -3267,7 +3267,7 @@ mod tests { fn streaming_raw() -> RawInputEvidence { let scope = streaming_scope(); - let node = PhysicalDagNode { + let node = PhysicalDAGNode { id: "raw-scan".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -3295,7 +3295,7 @@ mod tests { arriving_logical_row_bytes: 64, arriving_source_row_bytes: 64, ingestion_rate_per_second: 2.0, - physical_dag: EvidenceBackedPhysicalDag { + physical_dag: EvidenceBackedPhysicalDAG { nodes: vec![node], root: "raw-scan".into(), evidence: HashMap::from([( @@ -3330,7 +3330,7 @@ mod tests { SummaryExpr::KeepPreAsap(_) => { model.node_evidence.insert_retained_query( node, - RetainedSubDagEvidence { + RetainedSubDAGEvidence { physical_id: format!("retained-{node:p}"), output: test_edge(), preprocessing_cpu_ops_over_horizon: 1.0, diff --git a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs b/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs index 129d06ed3..8e63a4ce1 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_dag_export.rs @@ -1,6 +1,6 @@ //! Serializable DAG export for a materialized summary-maintenance plan. //! -//! `asap-types::dag_export` owns the crate-neutral post-ASAP graph shape. This +//! `asap-types::dag_export` owns the crate-neutral post-ASAP DAG shape. This //! adapter lives in the mapping layer, where summary-maintenance lifecycle //! alternatives and their typed rejection reasons are available, and emits //! both views together. @@ -10,7 +10,7 @@ use std::rc::Rc; use serde::Serialize; -use asap_types::dag_export::{self, SummaryDagGraph}; +use asap_types::dag_export::{self, SummaryDAG}; use asap_types::post_asap::{ PostAsapNodeId, ResultGuarantee, SummaryExpr, SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, SummaryNode, SummaryWindowFramework, @@ -21,8 +21,8 @@ use crate::summary_maintenance_lifecycle::{ }; #[derive(Debug, Clone, Serialize)] -pub struct SummaryMaintenanceDagExport { - pub graph: SummaryDagGraph, +pub struct SummaryMaintenanceDAGExport { + pub dag: SummaryDAG, pub deployments: Vec, pub horizon_seconds: Option, pub evaluation_rate_per_second: Option, @@ -63,7 +63,7 @@ pub type SummaryMaintenanceLifecycleGuaranteeExport = SummaryMaintenanceLifecycl pub fn export_summary_maintenance_plan( plan: &SummaryMaintenanceLifecyclePlan, -) -> SummaryMaintenanceDagExport { +) -> SummaryMaintenanceDAGExport { let deployments: Vec<_> = plan .deployments .iter() @@ -86,7 +86,7 @@ pub fn export_summary_maintenance_plan( .collect(), }) .collect(); - let mut graph = dag_export::export_summary(&plan.root); + let mut dag = dag_export::export_summary(&plan.root); let deployment_by_summary: HashMap<_, _> = plan .deployments .iter() @@ -96,13 +96,13 @@ pub fn export_summary_maintenance_plan( let mut next_node_id = 0; annotate_lifecycle_deployments( &plan.root, - &mut graph, + &mut dag, &deployment_by_summary, &mut next_node_id, ); - SummaryMaintenanceDagExport { - graph, + SummaryMaintenanceDAGExport { + dag, deployments, horizon_seconds: plan.horizon.map(|horizon| horizon.0), evaluation_rate_per_second: plan.evaluation_rate.map(|rate| rate.0), @@ -118,22 +118,22 @@ pub fn export_summary_maintenance_plan( /// Walk in the same post-order as `dag_export::export_summary` and attach a /// deployment directly to every flattened occurrence of its state node. -/// This makes the decision visible to graph consumers without asking them to -/// reconstruct pointer identity from graph position. +/// This makes the decision visible to DAG consumers without asking them to +/// reconstruct pointer identity from DAG position. fn annotate_lifecycle_deployments( node: &SummaryNode, - graph: &mut SummaryDagGraph, + dag: &mut SummaryDAG, deployments: &HashMap<*const SummaryNode, &SummaryMaintenanceDeploymentExport>, next_node_id: &mut usize, ) { if !matches!(node.expr, SummaryExpr::KeepPreAsap(_)) { for child in summary_children(&node.expr) { - annotate_lifecycle_deployments(child, graph, deployments, next_node_id); + annotate_lifecycle_deployments(child, dag, deployments, next_node_id); } } - let graph_node = &mut graph.nodes[*next_node_id]; + let dag_node = &mut dag.nodes[*next_node_id]; if let Some(deployment) = deployments.get(&(node as *const SummaryNode)) { - graph_node.detail["summary_maintenance"] = + dag_node.detail["summary_maintenance"] = serde_json::to_value(deployment).expect("lifecycle export is serializable"); } *next_node_id += 1; diff --git a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs index 6bc5b13c7..f0569d1ae 100644 --- a/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs +++ b/crates/asap-aware-mapping/src/summary_maintenance_lifecycle.rs @@ -21,9 +21,9 @@ use std::collections::{HashMap, HashSet}; use std::rc::Rc; use asap_types::post_asap::{ - compile_post_asap_dag_with_node_ids, share_common_summary_subtrees, EvaluationSchedule, - ExecutionDataStateError, ExecutionTiming, OutputRepresentation, PostAsapDag, - PostAsapDagValidationError, PostAsapNodeId, ResultGuarantee, SummaryExpr, + compile_post_asap_dag_with_node_ids, share_common_summary_sub_dags, EvaluationSchedule, + ExecutionDataStateError, ExecutionTiming, OutputRepresentation, PostAsapDAG, + PostAsapDAGValidationError, PostAsapNodeId, ResultGuarantee, SummaryExpr, SummaryMaintenanceLifecycle, SummaryMaintenanceLifecycleGuarantee, SummaryMaintenanceMode, SummaryNode, SummaryWindowFramework, ValueOperation, }; @@ -221,7 +221,7 @@ pub struct SummaryMaintenanceLifecyclePlan { #[derive(Debug, thiserror::Error, PartialEq)] pub enum SummaryMaintenanceTimingError { #[error(transparent)] - InvalidPostAsapDag(#[from] ExecutionDataStateError), + InvalidPostAsapDAG(#[from] ExecutionDataStateError), #[error("summary {0:?} has no selected lifecycle")] UnselectedLifecycle(PostAsapNodeId), /// A maintained population outside any `SummaryAgg`'s inputs has no @@ -230,7 +230,7 @@ pub enum SummaryMaintenanceTimingError { #[error("node {0:?} maintains state that has no summary-maintenance lifecycle")] UnplannedMaintainedState(PostAsapNodeId), #[error(transparent)] - InvalidPhases(#[from] PostAsapDagValidationError), + InvalidPhases(#[from] PostAsapDAGValidationError), } impl SummaryMaintenanceLifecyclePlan { @@ -245,7 +245,7 @@ impl SummaryMaintenanceLifecyclePlan { /// maintained populations as to `SummaryAgg` states; a population feeding /// a `SummaryAgg` is one of its inputs. Timings already on the root are /// ignored. - pub fn execution_timed_dag(&self) -> Result { + pub fn execution_timed_dag(&self) -> Result { let compiled = compile_post_asap_dag_with_node_ids(&self.root)?; let dag = compiled.dag; for population in &standalone_populations(&self.root) { @@ -355,13 +355,13 @@ pub enum SummaryMaintenanceLifecyclePlanError { #[error("workload entry index {index} appears more than once in one demand binding")] DuplicateWorkloadEntry { index: usize }, #[error(transparent)] - InvalidPostAsapDag(#[from] ExecutionDataStateError), + InvalidPostAsapDAG(#[from] ExecutionDataStateError), } #[derive(Debug, thiserror::Error)] pub enum SummaryMaintenanceLifecycleAssemblyError { #[error(transparent)] - AssembleDag(#[from] RealizationError), + AssembleDAG(#[from] RealizationError), #[error(transparent)] SummaryMaintenance(#[from] SummaryMaintenanceLifecyclePlanError), } @@ -800,7 +800,7 @@ pub fn global_selection_with_summary_maintenance_lifecycles<'a, Id>( // Intern every member once; members whose outermost state (the // `SummaryAgg` every other state of the candidate feeds) interns to the // same node share it. Classes are kept in first-member order. - let interned = share_common_summary_subtrees( + let interned = share_common_summary_sub_dags( members .iter() .enumerate() @@ -3306,7 +3306,7 @@ mod tests { data: &DataWorkload, horizon: Option, choose: impl Fn(&SummaryMaintenanceDeployment) -> SummaryMaintenanceLifecycle, - ) -> PostAsapDag { + ) -> PostAsapDAG { let candidates = enumerate_summary_maintenance_lifecycles( root, WorkloadDemand::new_with_data(workload, data, &[0]), @@ -3331,7 +3331,7 @@ mod tests { } /// Operator kinds in node-id order, each paired with its timing. - fn timings(dag: &PostAsapDag) -> Vec<(&'static str, ExecutionTiming)> { + fn timings(dag: &PostAsapDAG) -> Vec<(&'static str, ExecutionTiming)> { dag.nodes .iter() .map(|node| { @@ -3550,7 +3550,7 @@ mod tests { ) } - fn population_timings(dag: &PostAsapDag) -> Vec<(&'static str, ExecutionTiming)> { + fn population_timings(dag: &PostAsapDAG) -> Vec<(&'static str, ExecutionTiming)> { dag.nodes .iter() .zip(timings(dag)) diff --git a/crates/asap-aware-mapping/tests/physical_handoff_cost.rs b/crates/asap-aware-mapping/tests/physical_handoff_cost.rs index 0cf89169e..0885db71c 100644 --- a/crates/asap-aware-mapping/tests/physical_handoff_cost.rs +++ b/crates/asap-aware-mapping/tests/physical_handoff_cost.rs @@ -1,5 +1,5 @@ use asap_aware_mapping::analytical_cost::{ - EvidenceBackedPhysicalDag, ExecutionMultiplicity, PhysicalDagNode, PhysicalNodeEvidence, + EvidenceBackedPhysicalDAG, ExecutionMultiplicity, PhysicalDAGNode, PhysicalNodeEvidence, PhysicalOperator, }; use asap_aware_mapping::physical_operator_statistics::{ @@ -11,7 +11,7 @@ use asap_types::workload::{ }; use std::collections::HashMap; -fn fixture() -> (EvidenceBackedPhysicalDag, ComparisonScope) { +fn fixture() -> (EvidenceBackedPhysicalDAG, ComparisonScope) { let coverage = SourceCoverage { source: Source::Table { table_ref: "events".into(), @@ -45,7 +45,7 @@ fn fixture() -> (EvidenceBackedPhysicalDag, ComparisonScope) { promql: None, }; let nodes = vec![ - PhysicalDagNode { + PhysicalDAGNode { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -54,7 +54,7 @@ fn fixture() -> (EvidenceBackedPhysicalDag, ComparisonScope) { retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, }, - PhysicalDagNode { + PhysicalDAGNode { id: "left".into(), operator: PhysicalOperator::PassThrough, children: vec!["scan".into()], @@ -63,7 +63,7 @@ fn fixture() -> (EvidenceBackedPhysicalDag, ComparisonScope) { retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, }, - PhysicalDagNode { + PhysicalDAGNode { id: "right".into(), operator: PhysicalOperator::PassThrough, children: vec!["scan".into()], @@ -72,7 +72,7 @@ fn fixture() -> (EvidenceBackedPhysicalDag, ComparisonScope) { retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, }, - PhysicalDagNode { + PhysicalDAGNode { id: "root".into(), operator: PhysicalOperator::Concat, children: vec!["left".into(), "right".into()], @@ -127,7 +127,7 @@ fn fixture() -> (EvidenceBackedPhysicalDag, ComparisonScope) { ), ]); ( - EvidenceBackedPhysicalDag { + EvidenceBackedPhysicalDAG { nodes, root: "root".into(), evidence, @@ -161,7 +161,7 @@ fn mapping_resource_reexports_are_wire_compatible_shared_types() { ); } -fn profile(dag: &EvidenceBackedPhysicalDag) -> PhysicalHandoffProfile { +fn profile(dag: &EvidenceBackedPhysicalDAG) -> PhysicalHandoffProfile { PhysicalHandoffProfile { evidence_version: "evidence-v1".into(), observed_at_ms: 90, @@ -205,7 +205,7 @@ fn transfer(id: &str, consumer: Option<&str>) -> PhysicalHandoff { } } -fn profiles_for_alternatives(dags: &[&EvidenceBackedPhysicalDag]) -> PhysicalHandoffProfile { +fn profiles_for_alternatives(dags: &[&EvidenceBackedPhysicalDAG]) -> PhysicalHandoffProfile { let mut combined = profile(dags[0]); combined.plans.clear(); for dag in dags { diff --git a/crates/asap-aware-mapping/tests/storage_io.rs b/crates/asap-aware-mapping/tests/storage_io.rs index d8fd86143..a12118f09 100644 --- a/crates/asap-aware-mapping/tests/storage_io.rs +++ b/crates/asap-aware-mapping/tests/storage_io.rs @@ -1,5 +1,5 @@ use asap_aware_mapping::analytical_cost::{ - EvidenceBackedPhysicalDag, ExecutionMultiplicity, PhysicalDagNode, PhysicalNodeEvidence, + EvidenceBackedPhysicalDAG, ExecutionMultiplicity, PhysicalDAGNode, PhysicalNodeEvidence, PhysicalOperator, }; use asap_aware_mapping::physical_operator_statistics::{ @@ -11,7 +11,7 @@ use asap_types::workload::{ }; use std::collections::HashMap; -fn fixture() -> (EvidenceBackedPhysicalDag, ComparisonScope) { +fn fixture() -> (EvidenceBackedPhysicalDAG, ComparisonScope) { let coverage = SourceCoverage { source: Source::Table { table_ref: "events".into(), @@ -45,7 +45,7 @@ fn fixture() -> (EvidenceBackedPhysicalDag, ComparisonScope) { promql: None, }; let nodes = vec![ - PhysicalDagNode { + PhysicalDAGNode { id: "scan".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -54,7 +54,7 @@ fn fixture() -> (EvidenceBackedPhysicalDag, ComparisonScope) { retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, }, - PhysicalDagNode { + PhysicalDAGNode { id: "left".into(), operator: PhysicalOperator::PassThrough, children: vec!["scan".into()], @@ -63,7 +63,7 @@ fn fixture() -> (EvidenceBackedPhysicalDag, ComparisonScope) { retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, }, - PhysicalDagNode { + PhysicalDAGNode { id: "right".into(), operator: PhysicalOperator::PassThrough, children: vec!["scan".into()], @@ -72,7 +72,7 @@ fn fixture() -> (EvidenceBackedPhysicalDag, ComparisonScope) { retained_bytes: 0, execution: ExecutionMultiplicity::PerEvaluation, }, - PhysicalDagNode { + PhysicalDAGNode { id: "root".into(), operator: PhysicalOperator::Concat, children: vec!["left".into(), "right".into()], @@ -127,7 +127,7 @@ fn fixture() -> (EvidenceBackedPhysicalDag, ComparisonScope) { ), ]); ( - EvidenceBackedPhysicalDag { + EvidenceBackedPhysicalDAG { nodes, root: "root".into(), evidence, @@ -159,7 +159,7 @@ fn storage_estimates_use_the_shared_resource_type_without_wire_changes() { ); } -fn profile(dag: &EvidenceBackedPhysicalDag) -> StorageIoProfile { +fn profile(dag: &EvidenceBackedPhysicalDAG) -> StorageIoProfile { StorageIoProfile { evidence_version: "evidence-v1".into(), observed_at_ms: 90, diff --git a/crates/asap-physical-operators/README.md b/crates/asap-physical-operators/README.md index 5f6384659..2d3e7b569 100644 --- a/crates/asap-physical-operators/README.md +++ b/crates/asap-physical-operators/README.md @@ -5,7 +5,7 @@ query time execution. The library requires neither backend engine, a server, a storage implementation, Arrow nor DataFusion. DataFusion informed the design; it is not the execution framework. -`plan::PhysicalDag` binds typed operator inputs to node IDs. Each execution starts +`plan::PhysicalDAG` binds typed operator inputs to node IDs. Each execution starts one producer per reachable node, shares output batches among its consumers, and bounds buffering. Dropping one consumer does not cancel other consumers. A `RunContext` carries query or ingestion scope, cancellation and byte accounting. @@ -24,7 +24,7 @@ use asap_physical_operators::{ expressions::Expression, operators::Operator, values::Value, - plan::PhysicalDag, + plan::PhysicalDAG, runtime::{Limits, RunContext, Scope}, }; use asap_physical_operators::planner::pre_asap::DataType; @@ -34,7 +34,7 @@ let source = Operator::scalar(Value::Int64(7), DataType::Int64)?; let negate = Operator::project(source.schema(), vec![ ("value".into(), Expression::Negate(Box::new(Expression::Column(0)))), ])?; -let mut plan = PhysicalDag::default(); +let mut plan = PhysicalDAG::default(); plan.add(0, vec![], source)?; plan.add(1, vec![0], negate)?; let run = RunContext::new( @@ -47,7 +47,7 @@ assert!(matches!(batch.rows()[0][0], Value::Int64(-7))); # Ok::<(), asap_physical_operators::dag::Error>(()) ``` -`physical_planner::compile` accepts a logical Post-ASAP DAG (`PostAsapDag`) and typed input contracts. +`physical_planner::compile` accepts a logical Post-ASAP DAG (`PostAsapDAG`) and typed input contracts. The resulting candidate is instantiated with deployment readers after selection. It rejects unsupported operations and schema mismatches before starting a source. Implement `PhysicalOperator` for a deployment source, including asynchronous I/O; computation operators remain in @@ -72,7 +72,7 @@ See [the design](../../docs/design_docs/physical-planning-and-deployment.md). ## Module boundaries -- `plan`: immutable graph, operator interface, schemas and execution properties. +- `plan`: immutable DAG, operator interface, schemas and execution properties. - `runtime`: per-run streams, shared producers, memory reservations and cancellation. - `expressions`: scalar evaluation; typed builders and the Planner expression adapter. - `operators`: projection, filter, joins, aggregate/window, sort, limit and summary implementations. @@ -91,7 +91,7 @@ state; operators own grouping. A source must declare `Boundedness::Bounded` to feed a blocking operator. The default for a custom raw source is `Unknown`; query or ingestion scope alone -does not promise that its cursor ends. `PhysicalDag::properties` validates these +does not promise that its cursor ends. `PhysicalDAG::properties` validates these requirements before any source starts and returns boundedness and emission mode for every reachable node. The memory connector declares finite input. Custom physical sources expose the same facts through `PhysicalOperator::properties`. @@ -104,14 +104,14 @@ There is no spill or partitioned parallel execution in this implementation. ## Physical compilation and deployment inputs -`physical_planner::compile` accepts a Planner `PostAsapDag`, typed -`InputContract`s and output roots. It returns a reusable `CompiledPhysicalDag` +`physical_planner::compile` accepts a Planner `PostAsapDAG`, typed +`InputContract`s and output roots. It returns a reusable `CompiledPhysicalDAG` containing selected native operators and no live readers. Compilation validates schemas, input ordering, sharing and boundedness before deployment source access. -A deployment calls `CompiledPhysicalDag::instantiate` with exactly the declared +A deployment calls `CompiledPhysicalDAG::instantiate` with exactly the declared inputs. This checks source schemas and execution properties and constructs the -runnable graph without repeating logical lowering. The graph executes through +runnable DAG without repeating logical lowering. The DAG executes through the shared runtime with independent per-run state. Window coverage, revision and maintenance-policy admission remain deployment/planning contracts; this compiler does not discover storage or silently change a selected maintenance strategy. diff --git a/crates/asap-physical-operators/src/dag/mod.rs b/crates/asap-physical-operators/src/dag/mod.rs index c7837672a..74835cd98 100644 --- a/crates/asap-physical-operators/src/dag/mod.rs +++ b/crates/asap-physical-operators/src/dag/mod.rs @@ -1,5 +1,5 @@ //! Compatibility imports. New code should use plan, runtime, operators, physical_planner and sources directly. -pub use crate::plan::{NodeId, PhysicalDag, PhysicalOperator}; +pub use crate::plan::{NodeId, PhysicalDAG, PhysicalOperator}; pub use crate::runtime::batch_execution; pub use crate::runtime::{ Input, Limits, OutputStream, Reservation, RunContext, Scope, SharedValue, diff --git a/crates/asap-physical-operators/src/physical_planner/candidates.rs b/crates/asap-physical-operators/src/physical_planner/candidates.rs index 7cac8ef2b..0e6ee86ba 100644 --- a/crates/asap-physical-operators/src/physical_planner/candidates.rs +++ b/crates/asap-physical-operators/src/physical_planner/candidates.rs @@ -1,4 +1,4 @@ -//! Compile maintenance-selected frontiers without deployment-specific graph rewrites. +//! Compile maintenance-selected frontiers without deployment-specific DAG rewrites. use super::*; /// One computation realization; lifecycle/window/revision requirements accompany @@ -7,8 +7,8 @@ use super::*; #[derive(Clone, serde::Serialize, serde::Deserialize)] #[serde(try_from = "UncheckedPhysicalASAPDAG")] pub struct PhysicalASAPDAG { - pub precompute: Option, - pub query: CompiledPhysicalDag, + pub precompute: Option, + pub query: CompiledPhysicalDAG, pub materialized_outputs: BTreeMap, } @@ -21,7 +21,7 @@ pub struct PhysicalASAPDAG { /// contract used to build each output. This API never treats a result from a /// different window or revision as interchangeable merely because types match. pub fn compile_candidate( - dag: &PostAsapDag, + dag: &PostAsapDAG, inputs: BTreeMap, roots: &[NodeId], frontier: &[NodeId], @@ -34,7 +34,7 @@ pub fn compile_candidate( /// each query DAG once and derives every placement choice from that result. /// The candidate is identical to [`compile_candidate`] for the same frontier. pub fn cut_candidate( - compiled: &CompiledPhysicalDag, + compiled: &CompiledPhysicalDAG, frontier: &[NodeId], ) -> Result { if frontier.is_empty() { @@ -93,7 +93,7 @@ pub fn cut_candidate( /// That holds while timing-dependent lowering (an ingestion-time `Binary` /// aligns by value column) has the same timing at compile time as here. /// A query-time node feeding an ingestion-time node has no valid placement. -pub fn frontier_from_timing(dag: &PostAsapDag) -> Result, Error> { +pub fn frontier_from_timing(dag: &PostAsapDAG) -> Result, Error> { use planner_types::post_asap::ExecutionTiming::IngestionTime; let timing = dag .nodes @@ -127,7 +127,7 @@ pub fn frontier_from_timing(dag: &PostAsapDag) -> Result, Error> { /// and deployment feasibility are evaluated separately before cost selection. /// Exceeding the search budget returns an error, never a partial inventory. pub fn enumerate_frontiers( - dag: &PostAsapDag, + dag: &PostAsapDAG, inputs: &BTreeMap, roots: &[NodeId], max_candidates: usize, @@ -136,7 +136,7 @@ pub fn enumerate_frontiers( } fn enumerate_compiled_frontiers( - compiled: &CompiledPhysicalDag, + compiled: &CompiledPhysicalDAG, max_candidates: usize, ) -> Result>, Error> { if max_candidates == 0 { @@ -192,7 +192,7 @@ fn enumerate_compiled_frontiers( /// individual failures visible; do not substitute another computation on error. /// The DAG is lowered once; each frontier is a [`cut_candidate`] of it. pub fn compile_candidates( - dag: &PostAsapDag, + dag: &PostAsapDAG, inputs: BTreeMap, roots: &[NodeId], frontiers: &[Vec], @@ -272,8 +272,8 @@ pub fn select_candidate( #[derive(serde::Deserialize)] #[serde(deny_unknown_fields)] struct UncheckedPhysicalASAPDAG { - precompute: Option, - query: CompiledPhysicalDag, + precompute: Option, + query: CompiledPhysicalDAG, materialized_outputs: BTreeMap, } impl TryFrom for PhysicalASAPDAG { @@ -331,7 +331,7 @@ mod tests { use super::*; use planner_types::workload::*; - fn grouped_rate() -> (PostAsapDag, BTreeMap, NodeId) { + fn grouped_rate() -> (PostAsapDAG, BTreeMap, NodeId) { let workload = PlanningWorkload { query_workload: QueryWorkload { language: QueryLanguage::PromQL, @@ -399,9 +399,9 @@ mod tests { } fn with_timing( - dag: &PostAsapDag, - timing: impl Fn(&PostAsapDagNode) -> planner_types::post_asap::ExecutionTiming, - ) -> PostAsapDag { + dag: &PostAsapDAG, + timing: impl Fn(&PostAsapDAGNode) -> planner_types::post_asap::ExecutionTiming, + ) -> PostAsapDAG { let mut timed = dag.clone(); for node in &mut timed.nodes { node.output_state.timing = timing(node); @@ -413,7 +413,7 @@ mod tests { timed } - fn raw_input(dag: &PostAsapDag) -> BTreeMap { + fn raw_input(dag: &PostAsapDAG) -> BTreeMap { let raw = dag .nodes .iter() diff --git a/crates/asap-physical-operators/src/physical_planner/compiled.rs b/crates/asap-physical-operators/src/physical_planner/compiled.rs index af0bbe393..66f476c2d 100644 --- a/crates/asap-physical-operators/src/physical_planner/compiled.rs +++ b/crates/asap-physical-operators/src/physical_planner/compiled.rs @@ -35,22 +35,22 @@ enum Node { /// Selected native operators and input slots. Rebinding never repeats lowering. /// Serde is format-agnostic; deployments choose the encoding and its versioning. -/// Deserialization validates the graph before it is usable. +/// Deserialization validates the DAG before it is usable. #[derive(Clone, serde::Serialize, serde::Deserialize)] -#[serde(try_from = "UncheckedDag")] -pub struct CompiledPhysicalDag { +#[serde(try_from = "UncheckedDAG")] +pub struct CompiledPhysicalDAG { nodes: BTreeMap, roots: Vec, } #[derive(serde::Deserialize)] #[serde(deny_unknown_fields)] -struct UncheckedDag { +struct UncheckedDAG { nodes: BTreeMap, roots: Vec, } -impl TryFrom for CompiledPhysicalDag { +impl TryFrom for CompiledPhysicalDAG { type Error = Error; - fn try_from(dag: UncheckedDag) -> Result { + fn try_from(dag: UncheckedDAG) -> Result { let result = Self { nodes: dag.nodes, roots: dag.roots, @@ -60,10 +60,10 @@ impl TryFrom for CompiledPhysicalDag { } } -impl CompiledPhysicalDag { +impl CompiledPhysicalDAG { /// Link already-selected physical fragments without lowering operators again. /// Fragment keys and source keys share a namespace; repeated dependency IDs - /// therefore remain one producer in the composed graph. + /// therefore remain one producer in the composed DAG. pub fn compose( sources: BTreeMap, fragments: BTreeMap, Self)>, @@ -285,8 +285,8 @@ impl CompiledPhysicalDag { pub fn instantiate<'a>( &self, mut sources: BTreeMap>, - ) -> Result, Error> { - let mut graph = PhysicalDag::default(); + ) -> Result, Error> { + let mut dag = PhysicalDAG::default(); for (&id, node) in &self.nodes { match node { Node::Input(contract) => { @@ -305,7 +305,7 @@ impl CompiledPhysicalDag { "physical input {id} violates its compiled contract" ))); } - graph.add_boxed( + dag.add_boxed( id, vec![], Box::new(CheckedSource { @@ -315,15 +315,15 @@ impl CompiledPhysicalDag { )?; } Node::Operator { inputs, operator } => { - graph.add(id, inputs.clone(), operator.clone())?; + dag.add(id, inputs.clone(), operator.clone())?; } } } if !sources.is_empty() { return Err(invalid("unexpected physical input binding")); } - graph.validate(&self.roots)?; - Ok(graph) + dag.validate(&self.roots)?; + Ok(dag) } } impl PhysicalOperator for InputContract { diff --git a/crates/asap-physical-operators/src/physical_planner/mod.rs b/crates/asap-physical-operators/src/physical_planner/mod.rs index 72aed2720..3121ede05 100644 --- a/crates/asap-physical-operators/src/physical_planner/mod.rs +++ b/crates/asap-physical-operators/src/physical_planner/mod.rs @@ -4,13 +4,13 @@ use crate::operators::ReadoutQuery; use crate::summary_kernels::exact::ExactReadout; use crate::{ operators::{Expression, Operator, Reduction, SortKey}, - plan::{Boundedness, Emission, NodeId, PhysicalDag, PhysicalOperator, PlanProperties}, + plan::{Boundedness, Emission, NodeId, PhysicalDAG, PhysicalOperator, PlanProperties}, values::{Batch, Schema}, Error, }; use planner_types::{ post_asap::{ - ExactOperation, PostAsapDag, PostAsapDagNode, PostAsapOperatorPayload as Payload, + ExactOperation, PostAsapDAG, PostAsapDAGNode, PostAsapOperatorPayload as Payload, SketchQuery, SummaryFamilyType, SummaryInputExpr, ValueOperation, }, pre_asap::{ @@ -43,27 +43,27 @@ pub use candidates::{ }; mod compiled; -pub use compiled::{CompiledPhysicalDag, InputContract}; +pub use compiled::{CompiledPhysicalDAG, InputContract}; mod row_values; /// Compile computation without opening or retaining deployment readers. /// Input contracts identify explicit boundaries selected by maintenance planning. pub fn compile( - dag: &PostAsapDag, + dag: &PostAsapDAG, inputs: BTreeMap, roots: &[NodeId], -) -> Result { +) -> Result { compile_internal(dag, inputs, roots) } /// Convenience for callers that already resolved inputs. Lowering still uses /// only their contracts, and instantiation checks those contracts again. pub fn bind<'a>( - dag: &PostAsapDag, + dag: &PostAsapDAG, sources: BTreeMap>, roots: &[NodeId], -) -> Result, Error> { +) -> Result, Error> { let inputs = sources .iter() .map(|(&id, source)| (id, InputContract::from_source(source.as_ref()))) @@ -73,11 +73,11 @@ pub fn bind<'a>( /// Resolve raw scan connectors before invoking the reader-independent compiler. pub fn bind_with_data_sources<'a>( - dag: &PostAsapDag, + dag: &PostAsapDAG, mut sources: BTreeMap>, roots: &[NodeId], data_sources: &crate::sources::DataSources, -) -> Result, Error> { +) -> Result, Error> { // Only resolve scans reachable below the selected input boundaries. let mut pending = roots.to_vec(); let mut seen = BTreeSet::new(); @@ -114,7 +114,7 @@ thread_local! { } /// Helper operators are numbered from their Planner node alone, above the u32 -/// Planner ID range, so every boundary choice yields a subgraph of the same +/// Planner ID range, so every boundary choice yields a sub-DAG of the same /// lowering and candidate cuts need not renumber operators. A node lowering to /// several helpers takes consecutive indices below its base. fn helper_id(node: NodeId, index: u64) -> NodeId { @@ -123,10 +123,10 @@ fn helper_id(node: NodeId, index: u64) -> NodeId { } fn compile_internal( - dag: &PostAsapDag, + dag: &PostAsapDAG, mut sources: BTreeMap, roots: &[NodeId], -) -> Result { +) -> Result { preflight_depth(dag)?; dag.validate().map_err(|e| invalid(e.to_string()))?; let nodes = dag @@ -153,7 +153,7 @@ fn compile_internal( let consumer = u64::from(edge.consumer.0); if let ( Payload::Fallback { expression }, - Some(PostAsapDagNode { + Some(PostAsapDAGNode { payload: Payload::Binary { .. }, .. }), @@ -179,7 +179,7 @@ fn compile_internal( || promql_fallback::raw_series_owner(*id).is_some_and(|owner| { matches!( nodes.get(&owner), - Some(PostAsapDagNode { + Some(PostAsapDAGNode { payload: Payload::Fallback { .. }, .. }) @@ -210,7 +210,7 @@ fn compile_internal( } } } - let mut graph = CompiledPhysicalDag::new(roots.to_vec()); + let mut physical_dag = CompiledPhysicalDAG::new(roots.to_vec()); for id in ordered { let node = nodes[&id]; let mut auxiliary = helper_id(id, 0); @@ -220,7 +220,7 @@ fn compile_internal( if source.schema != output { return Err(invalid("frontier does not have the declared schema")); } - graph.add_input(id, source)?; + physical_dag.add_input(id, source)?; } else { #[cfg(test)] LOWERED_NODES.with(|count| count.set(count.get() + 1)); @@ -233,7 +233,7 @@ fn compile_internal( if schemas.iter().any(|s| s != &schemas[0]) { return Err(invalid("summary merge inputs have different schemas")); } - graph.add( + physical_dag.add( auxiliary, inputs, Operator::union(schemas[0].clone(), schemas.len())?, @@ -260,7 +260,7 @@ fn compile_internal( let slot = promql_fallback::raw_series_input(id, i); match sources.remove(&slot) { Some(contract) if &contract.schema == schema => { - graph.add_input(slot, contract)? + physical_dag.add_input(slot, contract)? } Some(_) => { return Err(invalid(format!( @@ -289,11 +289,11 @@ fn compile_internal( .collect::>() }; for (operator, inputs) in steps { - graph.add(auxiliary, resolve(inputs, &ids), operator)?; + physical_dag.add(auxiliary, resolve(inputs, &ids), operator)?; ids.push(auxiliary); auxiliary -= 1; } - graph.add( + physical_dag.add( id, resolve(last_inputs, &ids), last.with_output_schema(output)?, @@ -328,7 +328,7 @@ fn compile_internal( let value = named_column(input, &ColumnRef::SampleValue)?; let lookback = i64::try_from(spec.lookback_ms) .map_err(|_| invalid("current-series lookback overflows"))?; - graph.add( + physical_dag.add( id, inputs, Operator::current_series(input.clone(), identity, coordinate, value, lookback)? @@ -369,11 +369,11 @@ fn compile_internal( let last = chain.pop().expect("nonempty chain"); let mut inputs = inputs; for operator in chain { - graph.add(auxiliary, inputs, operator)?; + physical_dag.add(auxiliary, inputs, operator)?; inputs = vec![auxiliary]; auxiliary -= 1; } - graph.add(id, inputs, last.with_output_schema(output)?)?; + physical_dag.add(id, inputs, last.with_output_schema(output)?)?; continue; }; let groups = spec @@ -382,7 +382,7 @@ fn compile_internal( .map(|name| named_column(&input, &ColumnRef::Named(name.clone()))) .collect::, _>>()?; let value = named_column(&input, &ColumnRef::SampleValue)?; - graph.add( + physical_dag.add( auxiliary, inputs, Operator::sort( @@ -395,7 +395,7 @@ fn compile_internal( groups.clone(), )?, )?; - graph.add( + physical_dag.add( id, vec![auxiliary], Operator::limit(input, *k as u64, 0, groups)?.with_output_schema(output)?, @@ -454,8 +454,8 @@ fn compile_internal( groups, )?; let compact = build.schema(); - graph.add(auxiliary, inputs, build)?; - graph.add( + physical_dag.add(auxiliary, inputs, build)?; + physical_dag.add( id, vec![auxiliary], Operator::scope_timestamp(compact, output)?, @@ -490,8 +490,8 @@ fn compile_internal( let [l, r] = sides; let binary = Operator::series_binary(l, r, operator.clone(), scalars) .map_err(|error| invalid(format!("node {id}: {error}")))?; - graph.add(auxiliary, vec![], scalar)?; - graph.add(id, operands, binary.with_output_schema(output)?)?; + physical_dag.add(auxiliary, vec![], scalar)?; + physical_dag.add(id, operands, binary.with_output_schema(output)?)?; auxiliary -= 1; continue; } @@ -520,7 +520,7 @@ fn compile_internal( [scalar(&inputs[0]), scalar(&inputs[1])], ) .map_err(|error| invalid(format!("node {id}: {error}")))?; - graph.add(id, inputs, binary.with_output_schema(output)?)?; + physical_dag.add(id, inputs, binary.with_output_schema(output)?)?; continue; } } @@ -555,16 +555,16 @@ fn compile_internal( .collect(); let project = Operator::project(actual, columns)?.with_output_schema(output.clone())?; - graph.add(auxiliary, inputs, readout)?; + physical_dag.add(auxiliary, inputs, readout)?; if temporal_readout_drops_name(node) { - graph.add(auxiliary - 1, vec![auxiliary], project)?; - graph.add( + physical_dag.add(auxiliary - 1, vec![auxiliary], project)?; + physical_dag.add( id, vec![auxiliary - 1], Operator::series_without_name(output)?, )?; } else { - graph.add(id, vec![auxiliary], project)?; + physical_dag.add(id, vec![auxiliary], project)?; } auxiliary -= 1; continue; @@ -600,20 +600,20 @@ fn compile_internal( } } if temporal_readout_drops_name(node) { - graph.add(auxiliary, inputs, operator)?; - graph.add(id, vec![auxiliary], Operator::series_without_name(output)?)?; + physical_dag.add(auxiliary, inputs, operator)?; + physical_dag.add(id, vec![auxiliary], Operator::series_without_name(output)?)?; } else { - graph.add(id, inputs, operator)?; + physical_dag.add(id, inputs, operator)?; } } } - graph.validate()?; - Ok(graph) + physical_dag.validate()?; + Ok(physical_dag) } // Temporal summary readouts produce PromQL vectors, whose range functions drop // the metric name before matching/filtering. Stored state retains its full identity. -fn temporal_readout_drops_name(node: &PostAsapDagNode) -> bool { +fn temporal_readout_drops_name(node: &PostAsapDAGNode) -> bool { node.output_schema .fields .iter() @@ -634,14 +634,14 @@ fn temporal_readout_drops_name(node: &PostAsapDagNode) -> bool { /// Bind a Planner node against the schemas supplied by its deployment edges. /// This is the same checked path used by complete DAG binding. -pub fn compile_node(node: &PostAsapDagNode, inputs: &[Schema]) -> Result { +pub fn compile_node(node: &PostAsapDAGNode, inputs: &[Schema]) -> Result { for schema in inputs { crate::values::validate_schema(schema)?; } bind_operation(node, inputs)?.with_output_schema(Arc::new(node.output_schema.clone())) } -fn bind_operation(node: &PostAsapDagNode, inputs: &[Schema]) -> Result { +fn bind_operation(node: &PostAsapDAGNode, inputs: &[Schema]) -> Result { if let Payload::Binary { operator } = &node.payload { let [left, right] = inputs else { return Err(invalid("binary requires two inputs")); @@ -1058,7 +1058,7 @@ impl PhysicalOperator for CheckedSource<'_> { } // Bound recursion before invoking the upstream recursive provenance validator. -fn preflight_depth(dag: &PostAsapDag) -> Result<(), Error> { +fn preflight_depth(dag: &PostAsapDAG) -> Result<(), Error> { let mut remaining = dag .nodes .iter() diff --git a/crates/asap-physical-operators/src/physical_planner/precompute.rs b/crates/asap-physical-operators/src/physical_planner/precompute.rs index 0674ff4c2..bc76821d4 100644 --- a/crates/asap-physical-operators/src/physical_planner/precompute.rs +++ b/crates/asap-physical-operators/src/physical_planner/precompute.rs @@ -77,7 +77,7 @@ pub fn raw_sample_row( /// Input contract of a precompute boundary: raw sample rows for a raw time /// series scan, otherwise the stored population of its summary state. -pub fn boundary_schema(node: &PostAsapDagNode) -> Result { +pub fn boundary_schema(node: &PostAsapDAGNode) -> Result { let Payload::Fallback { expression } = &node.payload else { return source_schema(&node.output_schema); }; @@ -154,10 +154,10 @@ pub fn is_population_schema(schema: &Schema) -> bool { /// Compile a complete selected precompute sub-DAG. Inputs are already-computed /// state boundaries; the deployment supplies groups, panes and states, never operations. pub fn compile( - dag: &PostAsapDag, + dag: &PostAsapDAG, frontiers: &[NodeId], roots: &[NodeId], -) -> Result { +) -> Result { preflight_depth(dag)?; dag.validate().map_err(|e| invalid(e.to_string()))?; let nodes = dag @@ -226,7 +226,7 @@ pub fn compile( continue; } if node.output_state.timing != ExecutionTiming::IngestionTime { - return Err(invalid("precompute graph contains a query-time operation")); + return Err(invalid("precompute DAG contains a query-time operation")); } let inputs = dependencies.get(&id).cloned().unwrap_or_default(); let schemas = inputs @@ -238,18 +238,23 @@ pub fn compile( .ok_or_else(|| invalid("missing precompute input")) }) .collect::, _>>()?; - let graph = fragment( + let physical_dag = fragment( node, &schemas, &inputs.iter().map(|id| nodes[id]).collect::>(), )?; - outputs.insert(id, graph.output_contract(graph.roots()[0])?.schema); - fragments.insert(id, (inputs, graph)); + outputs.insert( + id, + physical_dag + .output_contract(physical_dag.roots()[0])? + .schema, + ); + fragments.insert(id, (inputs, physical_dag)); } - CompiledPhysicalDag::compose(sources, fragments, roots.to_vec()) + CompiledPhysicalDAG::compose(sources, fragments, roots.to_vec()) } -fn validate_value_output(node: &PostAsapDagNode) -> Result<(), Error> { +fn validate_value_output(node: &PostAsapDAGNode) -> Result<(), Error> { let schema = &node.output_schema; // Physical population rows already carry the complete identity in `$population`. // Typed logical plans may expose its opaque series-identity column as metadata. @@ -289,10 +294,10 @@ fn validate_value_output(node: &PostAsapDagNode) -> Result<(), Error> { } fn fragment( - node: &PostAsapDagNode, + node: &PostAsapDAGNode, schemas: &[Schema], - parents: &[&PostAsapDagNode], -) -> Result { + parents: &[&PostAsapDAGNode], +) -> Result { let sources = schemas .iter() .enumerate() @@ -553,7 +558,7 @@ fn fragment( )) } }; - CompiledPhysicalDag::from_operators(sources, operators, vec![root]) + CompiledPhysicalDAG::from_operators(sources, operators, vec![root]) } /// Resolve keyed item identities over raw sample rows: labels (absent labels diff --git a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs index 3d92f0278..7f6ac49c1 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_fallback.rs @@ -1,4 +1,4 @@ -//! Compile a retained PromQL subtree (`Fallback`) from its typed expression. +//! Compile a retained PromQL sub-DAG (`Fallback`) from its typed expression. //! The deployment supplies the raw series of each selector; the Planner //! computes selection, range functions, subqueries, matching and aggregation. use super::*; diff --git a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs index 802777c5d..fba6060d4 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_rows.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_rows.rs @@ -80,7 +80,7 @@ pub fn series_row( /// TopK result; ranking remains a native physical operator. pub fn compile_current_series_readout( selected: &Rc, -) -> Result { +) -> Result { use planner_types::post_asap::{ compile_post_asap_dag, maintained_population::PopulationReadout, SummaryField, }; @@ -194,7 +194,7 @@ pub fn compile_rate_ranking( ) -> Result< ( Rc, - CompiledPhysicalDag, + CompiledPhysicalDAG, ), Error, > { @@ -253,7 +253,7 @@ pub fn compile_rate_ranking( /// Rate readouts runs at ingestion time: fresh aggregate state per closed /// window. The input is the complete collection of per-series counter states. pub fn compile_fixed_window_rate_aggregation( - dag: &planner_types::post_asap::PostAsapDag, + dag: &planner_types::post_asap::PostAsapDAG, ) -> Result { use planner_types::post_asap::{ExactKind, ExecutionTiming, SketchAlgorithm}; let sources = dag diff --git a/crates/asap-physical-operators/src/physical_planner/promql_values.rs b/crates/asap-physical-operators/src/physical_planner/promql_values.rs index 98032505b..0f0bcf0b4 100644 --- a/crates/asap-physical-operators/src/physical_planner/promql_values.rs +++ b/crates/asap-physical-operators/src/physical_planner/promql_values.rs @@ -12,13 +12,13 @@ pub fn matrix_schema() -> Schema { crate::operators::vector_window::matrix_schema() } -pub fn compile_scalar(value: f64) -> Result { +pub fn compile_scalar(value: f64) -> Result { let operator = Operator::scalar( crate::values::Value::Float64(value), planner_types::pre_asap::DataType::Float64, )? .with_output_schema(scalar_schema())?; - CompiledPhysicalDag::from_operators( + CompiledPhysicalDAG::from_operators( BTreeMap::new(), BTreeMap::from([(0, (vec![], operator))]), vec![0], @@ -28,7 +28,7 @@ pub fn compile_scalar(value: f64) -> Result { pub fn compile_temporal( intent: &AggIntent, preserve_metric_name: bool, -) -> Result { +) -> Result { let operator = Operator::range_window(intent.clone())?; let mut operators = vec![operator]; if !preserve_metric_name { @@ -50,8 +50,8 @@ pub fn compile_temporal( unary(operators, matrix_schema()) } -pub fn compile_histogram_quantile() -> Result { - CompiledPhysicalDag::from_operators( +pub fn compile_histogram_quantile() -> Result { + CompiledPhysicalDAG::from_operators( BTreeMap::from([ (0, InputContract::bounded(scalar_schema())), (1, InputContract::bounded(vector_schema())), @@ -67,11 +67,11 @@ pub fn compile_binary( return_bool: bool, left_scalar: bool, right_scalar: bool, -) -> Result { +) -> Result { let left = crate::operators::vector_binary::value_schema(left_scalar); let right = crate::operators::vector_binary::value_schema(right_scalar); let op = Operator::vector_binary(left.clone(), right.clone(), operator.clone(), return_bool)?; - CompiledPhysicalDag::from_operators( + CompiledPhysicalDAG::from_operators( BTreeMap::from([ (0, InputContract::bounded(left)), (1, InputContract::bounded(right)), @@ -81,9 +81,9 @@ pub fn compile_binary( ) } -fn unary(operators: Vec, input: Schema) -> Result { +fn unary(operators: Vec, input: Schema) -> Result { let root = operators.len() as u64; - CompiledPhysicalDag::from_operators( + CompiledPhysicalDAG::from_operators( BTreeMap::from([(0, InputContract::bounded(input))]), operators .into_iter() @@ -134,7 +134,7 @@ fn vector_output(input: Schema, labels: usize, value: usize) -> Result, grouping: &GroupKeys, -) -> Result { +) -> Result { let project = grouped(grouping)?; let reduction = match intent { AggIntent::Sum { .. } => Reduction::Sum(1), @@ -153,7 +153,7 @@ pub fn compile_aggregate( pub fn compile_sort( descending: bool, grouping: &GroupKeys, -) -> Result { +) -> Result { let project = grouped(grouping)?; let sort = Operator::sort( project.schema(), @@ -172,14 +172,14 @@ pub fn compile_limit( n: u64, offset: u64, grouping: &GroupKeys, -) -> Result { +) -> Result { let project = grouped(grouping)?; let limit = Operator::limit(project.schema(), n, offset, vec![2])?; let output = vector_output(limit.schema(), 0, 1)?; unary(vec![project, limit, output], vector_schema()) } -pub fn compile_negate(scalar: bool) -> Result { +pub fn compile_negate(scalar: bool) -> Result { let input = if scalar { scalar_schema() } else { @@ -200,7 +200,7 @@ pub fn compile_negate(scalar: bool) -> Result { unary(vec![Operator::project(input.clone(), columns)?], input) } -pub fn compile_vector_to_scalar() -> Result { +pub fn compile_vector_to_scalar() -> Result { unary( vec![Operator::vector_to_scalar(vector_schema(), 1)?.with_output_schema(scalar_schema())?], vector_schema(), @@ -224,7 +224,7 @@ pub fn compile_exact_readout( family: SummaryFamilyType, lookback_ms: u64, preserve_metric_name: bool, -) -> Result { +) -> Result { use planner_types::post_asap::ExactKind; let statistic = match &family { SummaryFamilyType::ExactAggregate(kind, _) => match kind { diff --git a/crates/asap-physical-operators/src/plan/mod.rs b/crates/asap-physical-operators/src/plan/mod.rs index 0dce4dd8e..e54688a9e 100644 --- a/crates/asap-physical-operators/src/plan/mod.rs +++ b/crates/asap-physical-operators/src/plan/mod.rs @@ -1,4 +1,4 @@ -//! Immutable physical graph, operator contracts and pre-execution validation. +//! Immutable physical DAG, operator contracts and pre-execution validation. use crate::{ runtime::{Input, OutputStream, RunContext}, Error, @@ -42,17 +42,17 @@ pub(crate) struct Node<'a, V, S> { pub(crate) inputs: Vec, pub(crate) operator: Box + 'a>, } -pub struct PhysicalDag<'a, V, S> { +pub struct PhysicalDAG<'a, V, S> { pub(crate) nodes: BTreeMap>, } -impl Default for PhysicalDag<'_, V, S> { +impl Default for PhysicalDAG<'_, V, S> { fn default() -> Self { Self { nodes: BTreeMap::new(), } } } -impl<'a, V: 'a, S: Clone + PartialEq + Debug + 'a> PhysicalDag<'a, V, S> { +impl<'a, V: 'a, S: Clone + PartialEq + Debug + 'a> PhysicalDAG<'a, V, S> { pub fn add( &mut self, id: NodeId, @@ -79,7 +79,7 @@ impl<'a, V: 'a, S: Clone + PartialEq + Debug + 'a> PhysicalDag<'a, V, S> { /// Derive properties while checking topology and schemas, before starting sources. pub fn properties(&self, roots: &[NodeId]) -> Result, Error> { fn visit( - dag: &PhysicalDag<'_, V, S>, + dag: &PhysicalDAG<'_, V, S>, id: NodeId, active: &mut BTreeSet, done: &mut BTreeMap, diff --git a/crates/asap-physical-operators/src/runtime/batch_execution.rs b/crates/asap-physical-operators/src/runtime/batch_execution.rs index f63004401..3de02492b 100644 --- a/crates/asap-physical-operators/src/runtime/batch_execution.rs +++ b/crates/asap-physical-operators/src/runtime/batch_execution.rs @@ -2,7 +2,7 @@ //! bridge for deployments whose boundary values are not yet streaming batches. use crate::{ operators::Operator, - plan::PhysicalDag, + plan::PhysicalDAG, runtime::{RunContext, SharedValue}, values::Batch, Error, @@ -17,18 +17,18 @@ pub fn evaluate_batch( operators: Vec, context: RunContext, ) -> Result>, Error> { - let mut graph = PhysicalDag::default(); - graph.add( + let mut dag = PhysicalDAG::default(); + dag.add( 0, vec![], Operator::source(input.schema().clone(), vec![input])?, )?; let mut root = 0; for operator in operators { - graph.add(root + 1, vec![root], operator)?; + dag.add(root + 1, vec![root], operator)?; root += 1; } - evaluate_graph(graph, root, context) + evaluate_dag(dag, root, context) } /// Bind the ordered in-memory inputs of a native multi-input operator. @@ -37,17 +37,17 @@ pub fn evaluate_inputs( operator: Operator, context: RunContext, ) -> Result>, Error> { - let mut graph = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); let root = inputs.len() as u64; for (id, input) in inputs.into_iter().enumerate() { - graph.add( + dag.add( id as u64, vec![], Operator::source(input.schema().clone(), vec![input])?, )?; } - graph.add(root, (0..root).collect(), operator)?; - evaluate_graph(graph, root, context) + dag.add(root, (0..root).collect(), operator)?; + evaluate_dag(dag, root, context) } /// Evaluate a native in-memory source, including scalar sources, in the caller's scope. @@ -55,17 +55,17 @@ pub fn evaluate_source( source: Operator, context: RunContext, ) -> Result>, Error> { - let mut graph = PhysicalDag::default(); - graph.add(0, vec![], source)?; - evaluate_graph(graph, 0, context) + let mut dag = PhysicalDAG::default(); + dag.add(0, vec![], source)?; + evaluate_dag(dag, 0, context) } -fn evaluate_graph( - graph: PhysicalDag<'_, Batch, crate::values::Schema>, +fn evaluate_dag( + dag: PhysicalDAG<'_, Batch, crate::values::Schema>, root: crate::plan::NodeId, context: RunContext, ) -> Result>, Error> { - let mut output = graph.execute(&[root], context)?.remove(0); + let mut output = dag.execute(&[root], context)?.remove(0); let mut batches = Vec::new(); loop { match output.next().now_or_never() { diff --git a/crates/asap-physical-operators/src/runtime/mod.rs b/crates/asap-physical-operators/src/runtime/mod.rs index f72a57ef5..5498fee33 100644 --- a/crates/asap-physical-operators/src/runtime/mod.rs +++ b/crates/asap-physical-operators/src/runtime/mod.rs @@ -1,6 +1,6 @@ //! Per-run producer sharing, streams, backpressure and resource ownership. use crate::{ - plan::{NodeId, PhysicalDag}, + plan::{NodeId, PhysicalDAG}, Error, }; use futures::{stream::LocalBoxStream, Stream}; @@ -42,7 +42,7 @@ impl SharedValue { } pub(crate) fn execute<'r, V: 'r, S: Clone + PartialEq + Debug + 'r>( - dag: &'r PhysicalDag<'_, V, S>, + dag: &'r PhysicalDAG<'_, V, S>, roots: &[NodeId], context: RunContext, ) -> Result>, Error> { @@ -60,7 +60,7 @@ pub(crate) fn execute<'r, V: 'r, S: Clone + PartialEq + Debug + 'r>( } } fn build<'r, V: 'r, S: 'r>( - dag: &'r PhysicalDag<'_, V, S>, + dag: &'r PhysicalDAG<'_, V, S>, id: NodeId, context: &RunContext, states: &mut BTreeMap>>>, diff --git a/crates/asap-physical-operators/src/runtime/tests.rs b/crates/asap-physical-operators/src/runtime/tests.rs index f683041c6..3293edeba 100644 --- a/crates/asap-physical-operators/src/runtime/tests.rs +++ b/crates/asap-physical-operators/src/runtime/tests.rs @@ -93,7 +93,7 @@ fn source(fail: bool) -> (Source, Rc>, Rc>) { #[test] fn shared_source_backpressure_and_reader_drop() { let (source, starts, polls) = source(false); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add(0, vec![], source).unwrap(); let context = context(); let mut readers = dag.execute(&[0, 0], context.clone()).unwrap(); @@ -124,7 +124,7 @@ fn shared_source_backpressure_and_reader_drop() { #[test] fn diamond_and_run_isolation() { let (source, starts, polls) = source(false); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add(0, vec![], source).unwrap(); dag.add(1, vec![0], Identity).unwrap(); dag.add(2, vec![0], Identity).unwrap(); @@ -151,7 +151,7 @@ fn diamond_and_run_isolation() { #[test] fn broadcast_error_and_cancel() { let (source, _, polls) = source(true); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add(0, vec![], source).unwrap(); let mut outputs = dag.execute(&[0, 0], context()).unwrap(); let a = outputs.pop().unwrap(); @@ -177,7 +177,7 @@ fn broadcast_error_and_cancel() { #[test] fn retained_outputs_count_against_budget() { let (source, _, _) = source(false); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add(0, vec![], source).unwrap(); let run = RunContext::new( Scope::Query { @@ -203,20 +203,20 @@ fn retained_outputs_count_against_budget() { assert_eq!(run.retained_bytes(), 0); } -// Invalid graphs fail before even starting a source. +// Invalid DAGs fail before even starting a source. #[test] -fn invalid_graphs_do_not_start_sources() { +fn invalid_dags_do_not_start_sources() { let (source, starts, _) = source(false); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add(0, vec![], source).unwrap(); dag.add(1, vec![2], Identity).unwrap(); dag.add(2, vec![1], Identity).unwrap(); assert!(dag.execute(&[0, 1], context()).is_err()); assert_eq!(starts.get(), 0); - let mut missing = PhysicalDag::default(); + let mut missing = PhysicalDAG::default(); missing.add(1, vec![9], Identity).unwrap(); assert!(missing.validate(&[1]).is_err()); - let mut arity = PhysicalDag::default(); + let mut arity = PhysicalDAG::default(); arity.add(1, vec![], Identity).unwrap(); assert!(arity.validate(&[1]).is_err()); } @@ -226,7 +226,7 @@ fn invalid_graphs_do_not_start_sources() { fn ready_sources_cooperate_with_cancellation() { let (mut source, _, polls) = source(false); source.end = 10_000; - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add(0, vec![], source).unwrap(); let context = context(); let mut input = dag.execute(&[0], context.clone()).unwrap().remove(0); @@ -253,7 +253,7 @@ fn ready_sources_cooperate_with_cancellation() { #[test] fn depth_limit_covers_shared_paths() { let (source, _, _) = source(false); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add(0, vec![], source).unwrap(); for id in 1..129 { dag.add(id, vec![id - 1], Identity).unwrap(); diff --git a/crates/asap-physical-operators/tests/blocking_resources.rs b/crates/asap-physical-operators/tests/blocking_resources.rs index 7a312892b..27ddd1095 100644 --- a/crates/asap-physical-operators/tests/blocking_resources.rs +++ b/crates/asap-physical-operators/tests/blocking_resources.rs @@ -1,7 +1,7 @@ //! Blocking operators enforce resources before returning their first batch. use asap_physical_operators::{ operators::Operator, - plan::{PhysicalDag, PhysicalOperator}, + plan::{PhysicalDAG, PhysicalOperator}, runtime::{Limits, RunContext, Scope}, values::{Batch, Schema, Value}, Error, @@ -38,8 +38,8 @@ fn context(max_bytes: usize) -> RunContext { ) .unwrap() } -fn source(n: usize) -> PhysicalDag<'static, Batch, Schema> { - let mut dag = PhysicalDag::default(); +fn source(n: usize) -> PhysicalDAG<'static, Batch, Schema> { + let mut dag = PhysicalDAG::default(); dag.add( 0, vec![], @@ -150,7 +150,7 @@ fn cooperative_sort_preserves_ties_across_chunks() { .collect(), ) .unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add(0, vec![], Operator::source(schema(2), vec![batch]).unwrap()) .unwrap(); dag.add( @@ -201,7 +201,7 @@ fn weighted_summary_build_yields_within_a_batch() { ], time_index: None, }); - let mut sources = PhysicalDag::default(); + let mut sources = PhysicalDAG::default(); let batch = Batch::try_new( input.clone(), (0..1500) diff --git a/crates/asap-physical-operators/tests/current_series_heap.rs b/crates/asap-physical-operators/tests/current_series_heap.rs index 322bd6fee..df0c17b95 100644 --- a/crates/asap-physical-operators/tests/current_series_heap.rs +++ b/crates/asap-physical-operators/tests/current_series_heap.rs @@ -3,7 +3,7 @@ use asap_physical_operators::{ operators::Operator, physical_planner::{ promql_rows::{decode_series_identity, series_row, SERIES_IDENTITY_COLUMN}, - CompiledPhysicalDag, InputContract, Source, + CompiledPhysicalDAG, InputContract, Source, }, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, @@ -30,13 +30,13 @@ fn schema() -> Arc { time_index: Some(0), }) } -fn run(program: &CompiledPhysicalDag, data: Batch, end: i64) -> Result, String> { - let recovered = serde_json::from_slice::( +fn run(program: &CompiledPhysicalDAG, data: Batch, end: i64) -> Result, String> { + let recovered = serde_json::from_slice::( &serde_json::to_vec(&program).map_err(|e| e.to_string())?, ) .map_err(|e| e.to_string())?; let input_id = recovered.input_contracts().next().unwrap().0; - let graph = recovered + let dag = recovered .instantiate(BTreeMap::from([( input_id, Box::new(Operator::source(data.schema().clone(), vec![data]).unwrap()) as Source<'_>, @@ -51,7 +51,7 @@ fn run(program: &CompiledPhysicalDag, data: Batch, end: i64) -> Result Batch { .collect(); Batch::try_new(schema, rows).unwrap() } -fn snapshot_plan() -> CompiledPhysicalDag { - CompiledPhysicalDag::from_operators( +fn snapshot_plan() -> CompiledPhysicalDAG { + CompiledPhysicalDAG::from_operators( BTreeMap::from([(0, InputContract::bounded(schema()))]), BTreeMap::from([( 1, @@ -175,7 +175,7 @@ fn spatial_heap_ranks_latest_values_in_independent_runs() { time_index: None, }); let read = Operator::keyed_readout(build.schema(), 1, 1, output).unwrap(); - let plan = CompiledPhysicalDag::from_operators( + let plan = CompiledPhysicalDAG::from_operators( BTreeMap::from([(0, InputContract::bounded(schema()))]), BTreeMap::from([ ( @@ -230,7 +230,7 @@ fn current_series_observes_resource_limits() { let plan = snapshot_plan(); for cancelled in [false, true] { let data = input(&[("one", 50_000, 1.)]); - let graph = plan + let dag = plan .instantiate(BTreeMap::from([( 0, Box::new(Operator::source(data.schema().clone(), vec![data]).unwrap()) @@ -251,7 +251,7 @@ fn current_series_observes_resource_limits() { if cancelled { context.cancel(); } - let result = match graph.execute(&[1], context.clone()) { + let result = match dag.execute(&[1], context.clone()) { Err(error) => Err(error), Ok(mut streams) => block_on(streams.remove(0).next()).unwrap().map(|_| ()), }; diff --git a/crates/asap-physical-operators/tests/deployment_computation.rs b/crates/asap-physical-operators/tests/deployment_computation.rs index 3a92e4d96..a3738ac53 100644 --- a/crates/asap-physical-operators/tests/deployment_computation.rs +++ b/crates/asap-physical-operators/tests/deployment_computation.rs @@ -2,7 +2,7 @@ //! the deployment supplies only raw rows at the ingestion frontier. use asap_physical_operators::{ operators::Operator, - physical_planner::{compile, promql_rows, CompiledPhysicalDag, InputContract, Source}, + physical_planner::{compile, promql_rows, CompiledPhysicalDAG, InputContract, Source}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; @@ -45,7 +45,7 @@ fn lower_with(query: &str, accuracy: AccuracyTarget) -> QueryExpr { } /// The first exact summary candidate, as Planner selection would hand it over. -fn exact_dag(query: &str) -> PostAsapDag { +fn exact_dag(query: &str) -> PostAsapDAG { use asap_aware_mapping::{Replacement, ReplacementStrategy, TargetSubDAG}; let expression = lower(query); let root = Rc::new(promql_rows::with_series_identity(&expression).unwrap_or(expression)); @@ -65,7 +65,7 @@ fn exact_dag(query: &str) -> PostAsapDag { .unwrap() } -fn population_dag(query: &str) -> PostAsapDag { +fn population_dag(query: &str) -> PostAsapDAG { let root = Rc::new(promql_rows::with_series_identity(&lower(query)).unwrap()); let selected = asap_aware_mapping::maintained_population::MaintainedPopulationStrategy::new( std::slice::from_ref(&root), @@ -76,7 +76,7 @@ fn population_dag(query: &str) -> PostAsapDag { } /// Raw scan nodes are the frontier; everything above them is compiled. -fn raw_inputs(dag: &PostAsapDag) -> Vec<(u64, Arc, String)> { +fn raw_inputs(dag: &PostAsapDAG) -> Vec<(u64, Arc, String)> { dag.nodes .iter() .filter_map(|node| match &node.payload { @@ -103,7 +103,7 @@ type Sample = (&'static str, &'static str, &'static str, i64, f64); /// Compile, round-trip, bind raw `(metric, job, instance, ts, value)` samples, /// and return the root's batches. fn execute( - dag: &PostAsapDag, + dag: &PostAsapDAG, samples: &[Sample], end: i64, ) -> Result>, String> { @@ -113,7 +113,7 @@ fn execute( /// [`execute`], supplying samples of each instance in `relabel` under its /// `(__name__, instance)` instead. fn execute_relabeled( - dag: &PostAsapDag, + dag: &PostAsapDAG, samples: &[Sample], end: i64, relabel: &BTreeMap<&str, (&str, &str)>, @@ -128,7 +128,7 @@ fn execute_relabeled( &[u64::from(dag.root.0)], ) .map_err(|e| e.to_string())?; - let program: CompiledPhysicalDag = + let program: CompiledPhysicalDAG = serde_json::from_slice(&serde_json::to_vec(&program).unwrap()).unwrap(); let sources = inputs .iter() @@ -171,7 +171,7 @@ fn execute_relabeled( ) }) .collect(); - let graph = program.instantiate(sources).map_err(|e| e.to_string())?; + let physical_dag = program.instantiate(sources).map_err(|e| e.to_string())?; let context = RunContext::new( Scope::Query { evaluation_time_ms: end, @@ -181,7 +181,7 @@ fn execute_relabeled( ) .unwrap(); block_on(async { - let mut stream = graph + let mut stream = physical_dag .execute(program.roots(), context) .map_err(|e| e.to_string())? .remove(0); @@ -194,7 +194,7 @@ fn execute_relabeled( } /// [`execute`], returning `(job, value)` rows of the root. -fn run(dag: &PostAsapDag, samples: &[Sample], end: i64) -> Result, String> { +fn run(dag: &PostAsapDAG, samples: &[Sample], end: i64) -> Result, String> { let mut rows = BTreeMap::new(); for batch in execute(dag, samples, end)? { let job = batch.schema().fields.iter().position(|f| f.name == "job"); @@ -353,7 +353,7 @@ fn exact_count_finalizes_to_declared_float_value() { } /// `dag` with its Binary operator replaced by `kind`. -fn with_kind(mut dag: PostAsapDag, kind: planner_types::pre_asap::BinaryOpKind) -> PostAsapDag { +fn with_kind(mut dag: PostAsapDAG, kind: planner_types::pre_asap::BinaryOpKind) -> PostAsapDAG { for node in &mut dag.nodes { if let PostAsapOperatorPayload::Binary { operator } = &mut node.payload { operator.kind = kind.clone(); @@ -409,7 +409,7 @@ fn per_series_comparisons_filter_or_return_bool() { /// [`execute`], returning per-series `(identity, value)` rows of the root, /// with NaN-aware formatting for comparison. -fn run_series(dag: &PostAsapDag, samples: &[Sample], end: i64) -> Result { +fn run_series(dag: &PostAsapDAG, samples: &[Sample], end: i64) -> Result { let mut rows = BTreeMap::new(); for batch in execute(dag, samples, end)? { let schema = batch.schema(); @@ -544,10 +544,10 @@ fn per_series_scalar_arithmetic_rejects_label_sets_equal_without_the_name() { } fn with_vector_match( - mut dag: PostAsapDag, + mut dag: PostAsapDAG, kind: planner_types::pre_asap::VectorMatchKind, labels: &[&str], -) -> PostAsapDag { +) -> PostAsapDAG { for node in &mut dag.nodes { if let PostAsapOperatorPayload::Binary { operator } = &mut node.payload { operator.vector_match = Some(planner_types::pre_asap::VectorMatch { @@ -677,7 +677,7 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { &[u64::from(dag.root.0)], ) .unwrap(); - let program: CompiledPhysicalDag = + let program: CompiledPhysicalDAG = serde_json::from_slice(&serde_json::to_vec(&program).unwrap()).unwrap(); let mut sketch = CountMinSketchAccumulator::new(*depth as usize, *width as usize); sketch.inner.update("a", 3.0); @@ -694,7 +694,7 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { }) .collect(); let batch = Batch::try_new(schema.clone(), vec![row]).unwrap(); - let graph = program + let physical_dag = program .instantiate(BTreeMap::from([( u64::from(state.id.0), Box::new(Operator::source(schema, vec![batch]).unwrap()) as Source<'_>, @@ -709,7 +709,10 @@ fn stored_count_min_bare_count_compiles_to_a_readout() { ) .unwrap(); let values = block_on(async { - let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let mut stream = physical_dag + .execute(program.roots(), context) + .unwrap() + .remove(0); let mut values = Vec::new(); while let Some(batch) = stream.next().await { values.extend(batch.unwrap().rows().iter().map(|row| row[0].clone())); diff --git a/crates/asap-physical-operators/tests/physical_dag.rs b/crates/asap-physical-operators/tests/physical_dag.rs index 652880e04..8e8602216 100644 --- a/crates/asap-physical-operators/tests/physical_dag.rs +++ b/crates/asap-physical-operators/tests/physical_dag.rs @@ -3,7 +3,7 @@ use asap_physical_operators::{ dag::{ operators::{Expression, Operator, Reduction, SortKey}, values::{Batch, Schema, Value}, - Limits, PhysicalDag, RunContext, Scope, + Limits, PhysicalDAG, RunContext, Scope, }, Statistic, }; @@ -26,7 +26,7 @@ fn schema(fields: &[(&str, DataType, bool)]) -> Schema { time_index: None, }) } -fn run(dag: &PhysicalDag<'_, Batch, Schema>, root: u64, scope: Scope) -> Vec> { +fn run(dag: &PhysicalDAG<'_, Batch, Schema>, root: u64, scope: Scope) -> Vec> { let context = RunContext::new( scope, Limits { @@ -85,7 +85,7 @@ fn grouped_sort_limit_across_batches() { .unwrap() }) .collect(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add( 0, vec![], @@ -122,7 +122,7 @@ fn summary_construction_merge_and_readout_at_both_phases() { let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); let build = Operator::summary_build(schema.clone(), family, 0, None, vec![]).unwrap(); let state = build.schema(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add(0, vec![], Operator::source(schema, batches).unwrap()) .unwrap(); dag.add(1, vec![0], build).unwrap(); @@ -181,7 +181,7 @@ fn diamond_semijoin_preserves_left_values_and_multiplicity() { ), ) .unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add( 0, vec![], @@ -210,7 +210,7 @@ fn exact_integer_and_empty_extrema() { vec![("sum".into(), Reduction::Sum(0))], ) .unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); let value = 9_007_199_254_740_993; dag.add( 0, @@ -228,7 +228,7 @@ fn exact_integer_and_empty_extrema() { .unwrap(); dag.add(1, vec![0], aggregate).unwrap(); assert!(matches!(run(&dag,1,query())[0][0],Value::Int64(v) if v==value+2)); - let mut empty = PhysicalDag::default(); + let mut empty = PhysicalDAG::default(); empty .add(0, vec![], Operator::source(schema.clone(), vec![]).unwrap()) .unwrap(); @@ -255,7 +255,7 @@ fn scalar_negation_and_vector_conversion() { ) .unwrap(); let convert = Operator::vector_to_scalar(project.schema(), 0).unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add(0, vec![], scalar).unwrap(); dag.add(1, vec![0], project).unwrap(); dag.add(2, vec![1], convert).unwrap(); @@ -266,7 +266,7 @@ fn scalar_negation_and_vector_conversion() { Box::new(Expression::Column(0)), ); let filter = Operator::filter(scalar.schema(), predicate).unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add(0, vec![], scalar).unwrap(); dag.add(1, vec![0], filter).unwrap(); assert!(run(&dag, 1, query()).is_empty()); @@ -315,7 +315,7 @@ fn kll_raw_partial_and_precomputed_are_native_dags() { let build = Operator::summary_build(input.clone(), family, 0, None, vec![]).unwrap(); let state = build.schema(); let build_range = |start: u32, end: u32| { - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); let batch = Batch::try_new( input.clone(), (start..end) @@ -343,7 +343,7 @@ fn kll_raw_partial_and_precomputed_are_native_dags() { let prefix = build_range(0, 64); let complete = build_range(0, 128); let query_plan = |stored: Option>>, raw_start: Option| { - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); let mut states = vec![]; if let Some(rows) = stored { dag.add( @@ -432,7 +432,7 @@ fn exact_state_and_family_validation() { family: family.clone(), state: Arc::new(acc), }; - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add( 0, vec![], @@ -481,22 +481,22 @@ fn bind_post_asap_before_execution() { use asap_physical_operators::dag::planner::bind; use planner_types::{ post_asap::{ - EdgeRole, ExecutionDataState, GroupingEdgeCompatibility, PostAsapDag, PostAsapDagEdge, - PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload, ValueOperation, + EdgeRole, ExecutionDataState, GroupingEdgeCompatibility, PostAsapDAG, PostAsapDAGEdge, + PostAsapDAGNode, PostAsapNodeId, PostAsapOperatorPayload, ValueOperation, WindowEdgeCompatibility, }, pre_asap::{ArithmeticOpKind, ProjectItem, QueryExpr, ScalarValue}, }; use std::{collections::BTreeMap, rc::Rc}; let schema = schema(&[("value", DataType::Float64, false)]); - let node = |id, payload| PostAsapDagNode { + let node = |id, payload| PostAsapDAGNode { id: PostAsapNodeId(id), payload, output_state: ExecutionDataState::QUERY_ROWS, output_schema: (*schema).clone(), guarantee: None, }; - let mut dag = PostAsapDag { + let mut dag = PostAsapDAG { nodes: vec![ node( 0, @@ -521,7 +521,7 @@ fn bind_post_asap_before_execution() { }, ), ], - edges: vec![PostAsapDagEdge { + edges: vec![PostAsapDAGEdge { producer: PostAsapNodeId(0), consumer: PostAsapNodeId(1), role: EdgeRole::Input, @@ -580,7 +580,7 @@ fn empty_exact_count_is_an_integer_state_readout() { ), ) .unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add(0, vec![], Operator::source(input, vec![]).unwrap()) .unwrap(); dag.add(1, vec![0], build).unwrap(); @@ -594,7 +594,7 @@ fn source_batches_must_match_the_bound_schema() { use asap_physical_operators::dag::{self, PhysicalOperator}; use planner_types::{ post_asap::{ - ExecutionDataState, PostAsapDag, PostAsapDagNode, PostAsapNodeId, + ExecutionDataState, PostAsapDAG, PostAsapDAGNode, PostAsapNodeId, PostAsapOperatorPayload, }, pre_asap::QueryExpr, @@ -631,8 +631,8 @@ fn source_batches_must_match_the_bound_schema() { } let expected = schema(&[("value", DataType::Float64, false)]); let starts = Rc::new(Cell::new(0)); - let plan = PostAsapDag { - nodes: vec![PostAsapDagNode { + let plan = PostAsapDAG { + nodes: vec![PostAsapDAGNode { id: PostAsapNodeId(0), payload: PostAsapOperatorPayload::Fallback { expression: QueryExpr::promql_scalar(1.), @@ -663,7 +663,7 @@ fn source_batches_must_match_the_bound_schema() { #[test] fn extrema_preserve_numeric_values_in_the_presence_of_nan() { let input = schema(&[("v", DataType::Float64, false)]); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add( 0, vec![], @@ -712,14 +712,14 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { ("score", DataType::Float64, false), ]); let keys_schema = schema(&[("key", DataType::Utf8, false)]); - let node = |id, payload, schema: &Schema| PostAsapDagNode { + let node = |id, payload, schema: &Schema| PostAsapDAGNode { id: PostAsapNodeId(id), payload, output_schema: (**schema).clone(), output_state: ExecutionDataState::QUERY_ROWS, guarantee: None, }; - let edge = |producer, consumer, role, schema: &Schema| PostAsapDagEdge { + let edge = |producer, consumer, role, schema: &Schema| PostAsapDAGEdge { producer: PostAsapNodeId(producer), consumer: PostAsapNodeId(consumer), role, @@ -729,7 +729,7 @@ fn planner_semijoin_sort_limit_contract_at_both_phases() { window: WindowEdgeCompatibility::NotApplicable, }; let groups = GroupKeys::by(vec![0]); - let dag = PostAsapDag { + let dag = PostAsapDAG { nodes: vec![ node( 0, @@ -880,7 +880,7 @@ fn planner_expressions_preserve_collection_and_nullable_types() { )], ) .unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add( 0, vec![], @@ -954,7 +954,7 @@ fn native_relational_join_kinds_preserve_unmatched_rows() { ("right", DataType::Int64, true), ]) }; - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); for (id, rows) in [ ( 0, @@ -1077,7 +1077,7 @@ fn assert_weighted_rate_topk(count_sketch: bool) { ("score", DataType::Float64, false), ]); let readout = Operator::keyed_readout(build.schema(), 1, 8, output.clone()).unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add( 0, vec![], @@ -1131,10 +1131,10 @@ fn assert_weighted_rate_topk(count_sketch: bool) { #[test] fn grouped_temporal_schema_compiles_and_executes_topk() { use asap_physical_operators::physical_planner::{ - compile_node, CompiledPhysicalDag, InputContract, Source, + compile_node, CompiledPhysicalDAG, InputContract, Source, }; use planner_types::post_asap::{ - ExecutionDataState, PostAsapDagNode, PostAsapNodeId, PostAsapOperatorPayload, + ExecutionDataState, PostAsapDAGNode, PostAsapNodeId, PostAsapOperatorPayload, ValueOperation, }; use planner_types::pre_asap::{ @@ -1159,7 +1159,7 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { .map(|c| (c.name.as_str(), c.dtype.clone(), c.nullable)) .collect::>(), ); - let node = |id, operation| PostAsapDagNode { + let node = |id, operation| PostAsapDAGNode { id: PostAsapNodeId(id), payload: PostAsapOperatorPayload::Value { operation }, output_state: ExecutionDataState::QUERY_ROWS, @@ -1193,14 +1193,14 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { std::slice::from_ref(&input), ) .unwrap(); - let compiled = CompiledPhysicalDag::from_operators( + let compiled = CompiledPhysicalDAG::from_operators( [(0, InputContract::bounded(input.clone()))].into(), [(1, (vec![0], sort)), (2, (vec![1], limit))].into(), vec![2], ) .unwrap(); let recovered = - serde_json::from_slice::(&serde_json::to_vec(&compiled).unwrap()) + serde_json::from_slice::(&serde_json::to_vec(&compiled).unwrap()) .unwrap(); assert_eq!(recovered.row_source(2), Some(0)); assert_eq!(recovered.operator_name(2), Some("Limit")); @@ -1235,7 +1235,7 @@ fn grouped_temporal_schema_compiles_and_executes_topk() { #[test] fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { use asap_physical_operators::physical_planner::{ - compile_node, CompiledPhysicalDag, InputContract, Source, + compile_node, CompiledPhysicalDAG, InputContract, Source, }; use planner_types::{ post_asap::*, @@ -1244,7 +1244,7 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { use std::{collections::BTreeMap, rc::Rc}; let schema = schema(&[("key", DataType::Utf8, false)]); for certified in [false, true] { - let node = PostAsapDagNode { + let node = PostAsapDAGNode { id: PostAsapNodeId(2), output_schema: (*schema).clone(), output_state: ExecutionDataState::QUERY_ROWS, @@ -1266,7 +1266,7 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { }), }, }; - let graph = CompiledPhysicalDag::from_operators( + let dag = CompiledPhysicalDAG::from_operators( [ (0, InputContract::bounded(schema.clone())), (1, InputContract::bounded(schema.clone())), @@ -1283,11 +1283,10 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { vec![2], ) .unwrap(); - let graph = - serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) - .unwrap(); + let dag = serde_json::from_slice::(&serde_json::to_vec(&dag).unwrap()) + .unwrap(); assert_eq!( - graph.certified_pruning_keys(2), + dag.certified_pruning_keys(2), certified.then_some(&[(0, 0)][..]) ); for complete in [false, true] { @@ -1315,11 +1314,11 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { ) }) .collect::>(); - let bound = graph.instantiate(sources).unwrap(); + let bound = dag.instantiate(sources).unwrap(); let result = block_on(async { let mut stream = bound .execute( - graph.roots(), + dag.roots(), RunContext::new(query(), Limits::default()).unwrap(), ) .unwrap() @@ -1348,7 +1347,7 @@ fn certified_pruning_rejects_missing_authoritative_values_after_recovery() { #[test] fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { use asap_physical_operators::physical_planner::{ - compile_node, CompiledPhysicalDag, InputContract, Source, + compile_node, CompiledPhysicalDAG, InputContract, Source, }; use planner_types::{ post_asap::*, @@ -1360,7 +1359,7 @@ fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { ("time", DataType::Timestamp, false), ("value", DataType::Float64, false), ]); - let node = PostAsapDagNode { + let node = PostAsapDAGNode { id: PostAsapNodeId(2), output_schema: (*input).clone(), output_state: ExecutionDataState::INGESTION_ROWS, @@ -1374,7 +1373,7 @@ fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { }, }, }; - let program = CompiledPhysicalDag::from_operators( + let program = CompiledPhysicalDAG::from_operators( [ (0, InputContract::bounded(input.clone())), (1, InputContract::bounded(input.clone())), @@ -1392,7 +1391,7 @@ fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { ) .unwrap(); let program = - serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) + serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) .unwrap(); for (right, expected) in [ (vec![("b", 2, 3.), ("a", 1, 2.)], Some(vec![8., 17.])), @@ -1422,9 +1421,9 @@ fn compiled_ingestion_binary_preserves_alignment_and_rejects_missing_updates() { ) }) .collect::>(); - let graph = program.instantiate(sources).unwrap(); + let dag = program.instantiate(sources).unwrap(); let result = block_on(async { - let mut stream = graph + let mut stream = dag .execute( program.roots(), RunContext::new(query(), Limits::default()).unwrap(), diff --git a/crates/asap-physical-operators/tests/physical_plan_recovery.rs b/crates/asap-physical-operators/tests/physical_plan_recovery.rs index 820bc6092..1d3625154 100644 --- a/crates/asap-physical-operators/tests/physical_plan_recovery.rs +++ b/crates/asap-physical-operators/tests/physical_plan_recovery.rs @@ -2,7 +2,7 @@ //! Deployments choose the encoding; JSON is used here only as a test format. use asap_physical_operators::{ operators::{Operator, SortKey}, - physical_planner::{CompiledPhysicalDag, InputContract}, + physical_planner::{CompiledPhysicalDAG, InputContract}, }; use planner_types::{ post_asap::{SummaryFamilyType, SummaryField, SummarySchema}, @@ -10,7 +10,7 @@ use planner_types::{ }; use std::{collections::BTreeMap, sync::Arc}; -fn sorted() -> CompiledPhysicalDag { +fn sorted() -> CompiledPhysicalDAG { let schema = Arc::new(SummarySchema { fields: vec![SummaryField { name: "value".into(), @@ -19,7 +19,7 @@ fn sorted() -> CompiledPhysicalDag { }], time_index: None, }); - CompiledPhysicalDag::from_operators( + CompiledPhysicalDAG::from_operators( BTreeMap::from([(0, InputContract::bounded(schema.clone()))]), BTreeMap::from([( 1, @@ -45,7 +45,7 @@ fn sorted() -> CompiledPhysicalDag { #[test] fn recovery_retains_selected_operator_and_rejects_invalid_contracts() { let bytes = serde_json::to_vec(&sorted()).unwrap(); - let recovered = serde_json::from_slice::(&bytes).unwrap(); + let recovered = serde_json::from_slice::(&bytes).unwrap(); assert_eq!(serde_json::to_vec(&recovered).unwrap(), bytes); for mutation in ["column", "edge", "output"] { let mut wire: serde_json::Value = serde_json::from_slice(&bytes).unwrap(); @@ -62,7 +62,7 @@ fn recovery_retains_selected_operator_and_rejects_invalid_contracts() { _ => unreachable!(), } assert!( - serde_json::from_slice::(&serde_json::to_vec(&wire).unwrap()) + serde_json::from_slice::(&serde_json::to_vec(&wire).unwrap()) .is_err(), "accepted {mutation}" ); @@ -74,7 +74,7 @@ fn candidate_recovery_preserves_materialization_boundary() { use asap_physical_operators::physical_planner::PhysicalASAPDAG; let precompute = sorted(); let output = InputContract::bounded(precompute.output_contract(1).unwrap().schema); - let query = CompiledPhysicalDag::from_operators( + let query = CompiledPhysicalDAG::from_operators( BTreeMap::from([(1, output.clone())]), BTreeMap::from([( 2, diff --git a/crates/asap-physical-operators/tests/physical_semantics.rs b/crates/asap-physical-operators/tests/physical_semantics.rs index 6460cf8a3..cea0faccf 100644 --- a/crates/asap-physical-operators/tests/physical_semantics.rs +++ b/crates/asap-physical-operators/tests/physical_semantics.rs @@ -4,7 +4,7 @@ use asap_physical_operators::{ expressions::CompiledExpression, operators::{Expression, Operator, Reduction, SortKey}, - plan::PhysicalDag, + plan::PhysicalDAG, runtime::{Limits, RunContext, Scope}, values::{Batch, Schema, Value}, }; @@ -41,7 +41,7 @@ fn context() -> RunContext { ) .unwrap() } -fn collect(dag: &PhysicalDag<'_, Batch, Schema>, root: u64) -> Vec> { +fn collect(dag: &PhysicalDAG<'_, Batch, Schema>, root: u64) -> Vec> { let run = context(); let rows = block_on(async { let mut stream = dag.execute(&[root], run.clone()).unwrap().remove(0); @@ -55,7 +55,7 @@ fn collect(dag: &PhysicalDag<'_, Batch, Schema>, root: u64) -> Vec> { rows } fn unary(input: Schema, batches: Vec>>, op: Operator) -> Vec> { - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); let batches = batches .into_iter() .map(|rows| Batch::try_new(input.clone(), rows).unwrap()) @@ -93,7 +93,7 @@ fn join(left: Vec, right: Vec, kind: JoinKind, keyed: bool) -> Vec Operator::relational_join(input.clone(), input.clone(), kind, &eq_predicate(), output) .unwrap() }; - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); for (id, values) in [(0, left), (1, right)] { let batches = values .into_iter() @@ -360,7 +360,7 @@ fn global_extrema_bind_with_planner_derived_schema() { .unwrap(); let result = derived.columns[0].clone(); let output = schema(&[(&result.name, result.dtype, result.nullable)]); - let node = PostAsapDagNode { + let node = PostAsapDAGNode { id: PostAsapNodeId(1), payload: PostAsapOperatorPayload::Value { operation: ValueOperation::Exact(ExactOperation::Aggregate { @@ -421,7 +421,7 @@ fn planner_comparisons_handle_nan_without_execution_errors() { #[test] fn limit_branch_finishes_without_blocking_shared_sibling() { let input = schema(&[("v", DataType::Int64, false)]); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); let batches = (0..100) .map(|v| Batch::try_new(input.clone(), vec![vec![Value::Int64(v)]]).unwrap()) .collect(); @@ -543,7 +543,7 @@ fn kll_partial_merge_and_multiple_readouts_preserve_population() { SketchKind::new(SketchAlgorithm::Kll, SketchParams::Kll { k: 512 }), Default::default(), ); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); for (id, range) in [(0, 0..64), (1, 64..128), (2, 0..128)] { let rows = range.map(|n| vec![Value::Float64(n as f64)]).collect(); dag.add( @@ -627,7 +627,7 @@ fn zero_column_output_obeys_memory_limit() { use asap_physical_operators::Error; let input = schema(&[]); let batch = Batch::try_new(input.clone(), vec![vec![]; 200]).unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add(0, vec![], Operator::source(input, vec![batch]).unwrap()) .unwrap(); let run = RunContext::new( @@ -669,7 +669,7 @@ fn empty_exact_summary_extrema_agree_with_ordinary_aggregation() { ) .unwrap(); let state = build.schema(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add(0, vec![], Operator::source(input.clone(), vec![]).unwrap()) .unwrap(); dag.add(1, vec![0], build).unwrap(); diff --git a/crates/asap-physical-operators/tests/plan_properties.rs b/crates/asap-physical-operators/tests/plan_properties.rs index f7a468ab8..272079a90 100644 --- a/crates/asap-physical-operators/tests/plan_properties.rs +++ b/crates/asap-physical-operators/tests/plan_properties.rs @@ -1,7 +1,7 @@ //! Finite-input contracts are validated before source execution. use asap_physical_operators::{ operators::{Operator, SortKey}, - plan::{Boundedness, Emission, PhysicalDag}, + plan::{Boundedness, Emission, PhysicalDAG}, runtime::{Limits, OutputStream, RunContext, Scope}, sources::{DataSources, RawSource}, values::{Batch, Schema}, @@ -70,7 +70,7 @@ fn blocking_inputs_require_an_explicit_finite_source() { predicates: vec![], }) .unwrap(); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add(0, vec![], scan).unwrap(); dag.add( 1, diff --git a/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs b/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs index 2f8d5249f..367650df2 100644 --- a/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs +++ b/crates/asap-physical-operators/tests/planspace_series_identity_heap.rs @@ -96,11 +96,11 @@ fn lower(query: &str, accuracy: &AccuracyTarget) -> Rc { ) } -type Dag = Vec<(usize, Rc)>; +type InventoryDAG = Vec<(usize, Rc)>; /// Candidate DAGs for query 1 of a two-query workload, with and without /// whole-root proposals. Query 0 is a bystander that must not multiply them. -fn inventories(query: &str, accuracy: AccuracyTarget) -> (Vec, Vec) { +fn inventories(query: &str, accuracy: AccuracyTarget) -> (Vec, Vec) { let roots = vec![ ( 0, @@ -124,7 +124,7 @@ fn inventories(query: &str, accuracy: AccuracyTarget) -> (Vec, Vec) { (enumerate(&full), enumerate(&logical)) } -fn carries_identity(dag: &Dag) -> bool { +fn carries_identity(dag: &InventoryDAG) -> bool { dag.iter().any(|(_, root)| { compile_post_asap_dag(root) .unwrap() diff --git a/crates/asap-physical-operators/tests/precompute_candidates.rs b/crates/asap-physical-operators/tests/precompute_candidates.rs index 5c22ae669..6a6669fc7 100644 --- a/crates/asap-physical-operators/tests/precompute_candidates.rs +++ b/crates/asap-physical-operators/tests/precompute_candidates.rs @@ -5,7 +5,7 @@ use asap_physical_operators::{ operators::Operator, physical_planner::{ compile, compile_candidate, compile_candidates, cut_candidate, enumerate_frontiers, - select_candidate, CandidateCost, CompiledPhysicalDag, InputContract, PhysicalASAPDAG, + select_candidate, CandidateCost, CompiledPhysicalDAG, InputContract, PhysicalASAPDAG, Source, }, runtime::{Limits, RunContext, Scope}, @@ -52,7 +52,7 @@ fn grouped_rate_space() -> asap_aware_mapping::CandidateLogicalASAPDAGs<&'static search_workload(vec![("grouped-rate", root)]) } -fn grouped_rate() -> PostAsapDag { +fn grouped_rate() -> PostAsapDAG { let space = grouped_rate_space(); let selected = space .global_selection(&DefaultCostModel) @@ -61,7 +61,7 @@ fn grouped_rate() -> PostAsapDag { .unwrap(); compile_post_asap_dag(&selected).unwrap() } -fn run(plan: &CompiledPhysicalDag, inputs: BTreeMap, scope: Scope) -> Vec { +fn run(plan: &CompiledPhysicalDAG, inputs: BTreeMap, scope: Scope) -> Vec { let sources = inputs .into_iter() .map(|(id, batch)| { @@ -560,7 +560,7 @@ fn enumerated_grouped_rate_candidates_execute_numeric_query_outputs() { /// The per-frontier lowering used before compile-once cuts: each boundary /// choice lowers the precompute and query DAGs from the logical DAG again. fn recompiled_candidate( - dag: &PostAsapDag, + dag: &PostAsapDAG, inputs: &BTreeMap, roots: &[u64], frontier: &[u64], @@ -590,7 +590,7 @@ fn recompiled_candidate( } fn assert_cuts_match_recompilation( - dag: &PostAsapDag, + dag: &PostAsapDAG, inputs: BTreeMap, roots: &[u64], min_frontiers: usize, diff --git a/crates/asap-physical-operators/tests/precompute_population.rs b/crates/asap-physical-operators/tests/precompute_population.rs index fa124c686..2c9e5f764 100644 --- a/crates/asap-physical-operators/tests/precompute_population.rs +++ b/crates/asap-physical-operators/tests/precompute_population.rs @@ -1,8 +1,8 @@ -//! Persisted precompute graphs preserve group/window identity and execute state-to-state computation. +//! Persisted precompute DAGs preserve group/window identity and execute state-to-state computation. use asap_physical_operators::{ factory::create_planner_accumulator, operators::Operator, - physical_planner::{precompute, CompiledPhysicalDag, Source}, + physical_planner::{precompute, CompiledPhysicalDAG, Source}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, Statistic, @@ -44,14 +44,14 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { (SummaryInputExpr::Constant(1.), 4.), ] { let nodes = vec![ - PostAsapDagNode { + PostAsapDAGNode { id: PostAsapNodeId(0), payload: PostAsapOperatorPayload::SummaryMerge, output_state: ExecutionDataState::INGESTION_SUMMARY, output_schema: state_schema.clone(), guarantee: None, }, - PostAsapDagNode { + PostAsapDAGNode { id: PostAsapNodeId(1), payload: PostAsapOperatorPayload::Value { operation: ValueOperation::FinalizeExactAccumulator, @@ -60,7 +60,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { output_schema: value_schema.clone(), guarantee: None, }, - PostAsapDagNode { + PostAsapDAGNode { id: PostAsapNodeId(2), payload: PostAsapOperatorPayload::Binary { operator: BinaryOperator { @@ -74,7 +74,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { output_schema: value_schema.clone(), guarantee: None, }, - PostAsapDagNode { + PostAsapDAGNode { id: PostAsapNodeId(3), payload: PostAsapOperatorPayload::SummaryAgg { family: family.clone(), @@ -98,7 +98,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { (2, 3, EdgeRole::Input), ] .into_iter() - .map(|(producer, consumer, role)| PostAsapDagEdge { + .map(|(producer, consumer, role)| PostAsapDAGEdge { producer: PostAsapNodeId(producer), consumer: PostAsapNodeId(consumer), role, @@ -108,7 +108,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { window: WindowEdgeCompatibility::NotApplicable, }) .collect(); - let dag = PostAsapDag { + let dag = PostAsapDAG { nodes, edges, root: PostAsapNodeId(3), @@ -137,7 +137,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { ); let program = precompute::compile(&dag, &[0], &[3]).unwrap(); let program = - serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) + serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) .unwrap(); assert_eq!(program.input_contracts().count(), 1); for revision in [1, 2] { @@ -175,7 +175,7 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { Box::new(Operator::source(input.schema().clone(), vec![input]).unwrap()) as Source<'_>, )]); - let graph = program.instantiate(sources).unwrap(); + let physical_dag = program.instantiate(sources).unwrap(); let context = RunContext::new( Scope::Ingestion { window_start_ms: 0, @@ -186,7 +186,10 @@ fn finalized_shared_panes_rebuild_one_global_summary_after_recovery() { ) .unwrap(); let output = block_on(async { - let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let mut stream = physical_dag + .execute(program.roots(), context) + .unwrap() + .remove(0); let batch = stream.next().await.unwrap().unwrap(); assert!(stream.next().await.is_none()); batch @@ -221,12 +224,12 @@ fn logical_schema(family: SummaryFamilyType) -> SummarySchema { time_index: None, } } -fn state_graph( +fn state_dag( family: SummaryFamilyType, target: Option, merge: bool, -) -> CompiledPhysicalDag { - let mut nodes = vec![PostAsapDagNode { +) -> CompiledPhysicalDAG { + let mut nodes = vec![PostAsapDAGNode { id: PostAsapNodeId(0), payload: PostAsapOperatorPayload::SummaryMerge, output_state: ExecutionDataState::INGESTION_SUMMARY, @@ -234,14 +237,14 @@ fn state_graph( guarantee: None, }]; if merge { - nodes.push(PostAsapDagNode { + nodes.push(PostAsapDAGNode { id: PostAsapNodeId(1), payload: PostAsapOperatorPayload::SummaryMerge, ..nodes[0].clone() }); } let read_id = nodes.len() as u32; - nodes.push(PostAsapDagNode { + nodes.push(PostAsapDAGNode { id: PostAsapNodeId(read_id), payload: PostAsapOperatorPayload::Value { operation: ValueOperation::FinalizeExactAccumulator, @@ -251,7 +254,7 @@ fn state_graph( guarantee: None, }); if let Some(target) = target { - nodes.push(PostAsapDagNode { + nodes.push(PostAsapDAGNode { id: PostAsapNodeId(nodes.len() as u32), payload: PostAsapOperatorPayload::SummaryAgg { family: target.clone(), @@ -266,7 +269,7 @@ fn state_graph( }); } let edges = (1..nodes.len()) - .map(|i| PostAsapDagEdge { + .map(|i| PostAsapDAGEdge { producer: nodes[i - 1].id, consumer: nodes[i].id, role: EdgeRole::Input, @@ -278,20 +281,20 @@ fn state_graph( .collect(); let root = nodes.last().unwrap().id; precompute::compile( - &PostAsapDag { nodes, edges, root }, + &PostAsapDAG { nodes, edges, root }, &[0], &[u64::from(root.0)], ) .unwrap() } fn native_run( - program: &CompiledPhysicalDag, + program: &CompiledPhysicalDAG, family: SummaryFamilyType, states: Vec>, context: RunContext, ) -> Result>, asap_physical_operators::Error> { let program = - serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) + serde_json::from_slice::(&serde_json::to_vec(&program).unwrap()) .unwrap(); let rows = states .into_iter() @@ -308,13 +311,13 @@ fn native_run( }) .collect(); let input = Batch::try_new(precompute::population_schema(family), rows)?; - let graph = program.instantiate(BTreeMap::from([( + let physical_dag = program.instantiate(BTreeMap::from([( 0, Box::new(Operator::source(input.schema().clone(), vec![input])?) as Source<'_>, )]))?; block_on(async { let mut rows = Vec::new(); - let mut stream = graph.execute(program.roots(), context)?.remove(0); + let mut stream = physical_dag.execute(program.roots(), context)?.remove(0); while let Some(batch) = stream.next().await { rows.extend(batch?.rows().iter().cloned()); } @@ -350,7 +353,7 @@ fn sum_state(value: f64) -> Arc { fn explicit_merge_changes_pane_cardinality() { let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); for (merge, expected) in [(false, vec![2., 7.]), (true, vec![9.])] { - let program = state_graph(family.clone(), None, merge); + let program = state_dag(family.clone(), None, merge); let rows = native_run( &program, family.clone(), @@ -381,7 +384,7 @@ fn precompute_rejects_nonfinite_and_nonpositive_dds_updates() { ), GroupingStrategy::default(), ); - let program = state_graph(source.clone(), Some(target), false); + let program = state_dag(source.clone(), Some(target), false); assert!(native_run( &program, source.clone(), @@ -405,7 +408,7 @@ fn precompute_rejects_nonfinite_and_nonpositive_dds_updates() { fn precompute_count_conversion_checks_precision() { use asap_physical_operators::summary_kernels::exact::ExactAccumulator; let family = SummaryFamilyType::ExactAggregate(ExactKind::Count, ExactParams::Count); - let program = state_graph(family.clone(), None, false); + let program = state_dag(family.clone(), None, false); for (count, valid) in [(3u64, true), ((1u64 << 53) + 1, false)] { let mut state = serde_json::to_value(ExactAccumulator::new(family.clone(), false).unwrap()).unwrap(); @@ -421,11 +424,11 @@ fn precompute_count_conversion_checks_precision() { } } -// Graph execution retains terminal cancellation and shared workspace limits. +// DAG execution retains terminal cancellation and shared workspace limits. #[test] -fn precompute_graph_enforces_cancellation_and_budget() { +fn precompute_dag_enforces_cancellation_and_budget() { let family = SummaryFamilyType::ExactAggregate(ExactKind::Sum, ExactParams::Sum); - let program = state_graph(family.clone(), None, true); + let program = state_dag(family.clone(), None, true); let context = ingestion_context(Limits::default()); context.cancel(); let error = native_run(&program, family.clone(), vec![sum_state(1.)], context).unwrap_err(); diff --git a/crates/asap-physical-operators/tests/promql_binary.rs b/crates/asap-physical-operators/tests/promql_binary.rs index 822bde4ea..bbfc62c0d 100644 --- a/crates/asap-physical-operators/tests/promql_binary.rs +++ b/crates/asap-physical-operators/tests/promql_binary.rs @@ -1,14 +1,14 @@ //! Binary computation must be fully compiled before deployment binds values. use asap_physical_operators::{ operators::Operator, - physical_planner::{compile_node, CompiledPhysicalDag, InputContract, Source}, + physical_planner::{compile_node, CompiledPhysicalDAG, InputContract, Source}, runtime::{Limits, RunContext, Scope}, values::{Batch, Schema, Value}, }; use futures::{executor::block_on, StreamExt}; use planner_types::{ post_asap::{ - BinaryOperator, ExecutionDataState, PostAsapDagNode, PostAsapNodeId, + BinaryOperator, ExecutionDataState, PostAsapDAGNode, PostAsapNodeId, PostAsapOperatorPayload, SummaryFamilyType, SummaryField, SummarySchema, }, pre_asap::{ArithmeticOpKind, BinaryOpKind, DataType}, @@ -48,7 +48,7 @@ fn row(name: &str, job: &str, value: f64) -> Vec { Value::Float64(value), ] } -fn program() -> CompiledPhysicalDag { +fn program() -> CompiledPhysicalDAG { program_for(BinaryOperator { kind: BinaryOpKind::Arithmetic(ArithmeticOpKind::Div), vector_match: None, @@ -56,9 +56,9 @@ fn program() -> CompiledPhysicalDag { checked_finite_division: false, }) } -fn program_for(operator: BinaryOperator) -> CompiledPhysicalDag { +fn program_for(operator: BinaryOperator) -> CompiledPhysicalDAG { let schema = schema(); - let node = PostAsapDagNode { + let node = PostAsapDAGNode { id: PostAsapNodeId(2), payload: PostAsapOperatorPayload::Binary { operator }, output_state: ExecutionDataState::QUERY_ROWS, @@ -66,7 +66,7 @@ fn program_for(operator: BinaryOperator) -> CompiledPhysicalDag { guarantee: None, }; let operator = compile_node(&node, &[schema.clone(), schema.clone()]).unwrap(); - let graph = CompiledPhysicalDag::from_operators( + let physical_dag = CompiledPhysicalDAG::from_operators( BTreeMap::from([ (0, InputContract::bounded(schema.clone())), (1, InputContract::bounded(schema)), @@ -75,7 +75,8 @@ fn program_for(operator: BinaryOperator) -> CompiledPhysicalDag { vec![2], ) .unwrap(); - serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()).unwrap() + serde_json::from_slice::(&serde_json::to_vec(&physical_dag).unwrap()) + .unwrap() } fn evaluate( left: Vec>, @@ -84,7 +85,7 @@ fn evaluate( evaluate_with(program(), left, right) } fn evaluate_with( - graph: CompiledPhysicalDag, + physical_dag: CompiledPhysicalDAG, left: Vec>, right: Vec>, ) -> Result>, asap_physical_operators::Error> { @@ -99,7 +100,7 @@ fn evaluate_with( ) }) .collect(); - let bound = graph.instantiate(sources)?; + let bound = physical_dag.instantiate(sources)?; let ctx = RunContext::new( Scope::Query { evaluation_time_ms: 1, @@ -150,7 +151,7 @@ fn scalar_broadcast_and_bool_comparison_are_distinct() { use asap_physical_operators::physical_planner::promql_values; use planner_types::pre_asap::CompareOpKind; for return_bool in [false, true] { - let graph = promql_values::compile_binary( + let physical_dag = promql_values::compile_binary( &BinaryOperator { kind: BinaryOpKind::Compare(CompareOpKind::Lt), vector_match: None, @@ -162,9 +163,10 @@ fn scalar_broadcast_and_bool_comparison_are_distinct() { false, ) .unwrap(); - let graph = - serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) - .unwrap(); + let physical_dag = serde_json::from_slice::( + &serde_json::to_vec(&physical_dag).unwrap(), + ) + .unwrap(); let scalar = promql_values::scalar_schema(); let vector = promql_values::vector_schema(); let sources = BTreeMap::from([ @@ -193,7 +195,7 @@ fn scalar_broadcast_and_bool_comparison_are_distinct() { ) as Source<'_>, ), ]); - let bound = graph.instantiate(sources).unwrap(); + let bound = physical_dag.instantiate(sources).unwrap(); let context = RunContext::new( Scope::Query { evaluation_time_ms: 1, @@ -232,7 +234,7 @@ fn scalar_broadcast_and_bool_comparison_are_distinct() { #[test] fn binary_obeys_memory_and_cancellation() { for cancel in [false, true] { - let graph = program(); + let physical_dag = program(); let sources = (0..2) .map(|id| { ( @@ -247,7 +249,7 @@ fn binary_obeys_memory_and_cancellation() { ) }) .collect(); - let bound = graph.instantiate(sources).unwrap(); + let bound = physical_dag.instantiate(sources).unwrap(); let context = RunContext::new( Scope::Query { evaluation_time_ms: 1, @@ -341,7 +343,7 @@ fn stored_series_readouts_support_filters_and_sets() { BinaryOpKind::Set(PromQLVectorSetOpKind::Or), ] { let nodes = (0..5) - .map(|id| PostAsapDagNode { + .map(|id| PostAsapDAGNode { id: PostAsapNodeId(id), payload: match id { 0 | 1 => PostAsapOperatorPayload::SummaryMerge, @@ -377,7 +379,7 @@ fn stored_series_readouts_support_filters_and_sets() { (3, 4, EdgeRole::Right), ] .into_iter() - .map(|(producer, consumer, role)| PostAsapDagEdge { + .map(|(producer, consumer, role)| PostAsapDAGEdge { producer: PostAsapNodeId(producer), consumer: PostAsapNodeId(consumer), role, @@ -387,12 +389,12 @@ fn stored_series_readouts_support_filters_and_sets() { window: WindowEdgeCompatibility::NotApplicable, }) .collect(); - let dag = PostAsapDag { + let dag = PostAsapDAG { nodes, edges, root: PostAsapNodeId(4), }; - let graph = compile( + let physical_dag = compile( &dag, BTreeMap::from([ (0, InputContract::bounded(state_schema.clone())), @@ -401,8 +403,8 @@ fn stored_series_readouts_support_filters_and_sets() { &[4], ) .unwrap(); - let graph: CompiledPhysicalDag = - serde_json::from_slice(&serde_json::to_vec(&graph).unwrap()).unwrap(); + let physical_dag: CompiledPhysicalDAG = + serde_json::from_slice(&serde_json::to_vec(&physical_dag).unwrap()).unwrap(); let sources = [(0, "a", 6.), (1, "b", 2.)] .into_iter() .map(|(id, name, value)| { @@ -437,7 +439,7 @@ fn stored_series_readouts_support_filters_and_sets() { ) }) .collect(); - let bound = graph.instantiate(sources).unwrap(); + let bound = physical_dag.instantiate(sources).unwrap(); let context = RunContext::new( Scope::Query { evaluation_time_ms: 1, diff --git a/crates/asap-physical-operators/tests/promql_fallback.rs b/crates/asap-physical-operators/tests/promql_fallback.rs index 88fe3690f..45b3526f6 100644 --- a/crates/asap-physical-operators/tests/promql_fallback.rs +++ b/crates/asap-physical-operators/tests/promql_fallback.rs @@ -1,9 +1,9 @@ -//! A retained PromQL subtree (`Fallback`) compiles from its typed expression. +//! A retained PromQL sub-DAG (`Fallback`) compiles from its typed expression. //! The deployment supplies only its selector's raw series; expected values are //! hand-computed with Prometheus semantics. use asap_physical_operators::{ operators::Operator, - physical_planner::{compile, promql_fallback, promql_rows, CompiledPhysicalDag, InputContract}, + physical_planner::{compile, promql_fallback, promql_rows, CompiledPhysicalDAG, InputContract}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; @@ -56,7 +56,7 @@ fn lower(query: &str) -> QueryExpr { } /// The whole query retained as one pre-ASAP node. -fn fallback_dag(expression: QueryExpr) -> PostAsapDag { +fn fallback_dag(expression: QueryExpr) -> PostAsapDAG { let schema = lift_plain(&expression.output_schema().unwrap()); compile_post_asap_dag(&Rc::new(SummaryNode { expr: SummaryExpr::KeepPreAsap(Rc::new(expression)), @@ -93,13 +93,13 @@ fn metric(selector: &QueryExpr) -> String { } } -fn compile_query(query: &str) -> Result { +fn compile_query(query: &str) -> Result { let expression = lower(query); compile_dag(&expression, &fallback_dag(expression.clone())) } /// Compile a DAG whose root is the Fallback computing `expression`. -fn compile_dag(expression: &QueryExpr, dag: &PostAsapDag) -> Result { +fn compile_dag(expression: &QueryExpr, dag: &PostAsapDAG) -> Result { let root = u64::from(dag.root.0); let inputs = promql_fallback::raw_series(expression) .map_err(|e| e.to_string())? @@ -131,7 +131,7 @@ fn evaluate( #[allow(clippy::type_complexity)] fn evaluate_dag( expression: &QueryExpr, - dag: &PostAsapDag, + dag: &PostAsapDAG, metrics: &[(&str, &[Sample])], at: i64, ) -> Result, i64, f64)>, String> { @@ -141,7 +141,7 @@ fn evaluate_dag( #[allow(clippy::type_complexity)] fn evaluate_dag_with_range( expression: &QueryExpr, - dag: &PostAsapDag, + dag: &PostAsapDAG, metrics: &[(&str, &[Sample])], at: i64, bounds: Option<(i64, i64)>, @@ -168,7 +168,7 @@ fn evaluate_dag_with_range( Box::new(Operator::source(schema, vec![batch]).unwrap()) as _, ); } - let graph = program.instantiate(sources).map_err(|e| e.to_string())?; + let physical_dag = program.instantiate(sources).map_err(|e| e.to_string())?; let context = RunContext::new( Scope::Query { evaluation_time_ms: at * 1000, @@ -184,7 +184,7 @@ fn evaluate_dag_with_range( None => context, }; block_on(async { - let mut stream = graph + let mut stream = physical_dag .execute(program.roots(), context) .map_err(|e| e.to_string())? .remove(0); @@ -446,14 +446,14 @@ fn raw_series_contract_is_explicit() { // turned into instant selection. let selector = lower("m"); let schema = lift_plain(&selector.output_schema().unwrap()); - let node = |id, payload| PostAsapDagNode { + let node = |id, payload| PostAsapDAGNode { id: PostAsapNodeId(id), payload, output_state: ExecutionDataState::QUERY_ROWS, output_schema: schema.clone(), guarantee: None, }; - let consumed = PostAsapDag { + let consumed = PostAsapDAG { nodes: vec![ node( 0, @@ -472,7 +472,7 @@ fn raw_series_contract_is_explicit() { }, ), ], - edges: vec![PostAsapDagEdge { + edges: vec![PostAsapDAGEdge { producer: PostAsapNodeId(0), consumer: PostAsapNodeId(1), role: EdgeRole::Input, diff --git a/crates/asap-physical-operators/tests/promql_values.rs b/crates/asap-physical-operators/tests/promql_values.rs index 886b7d00d..108d9b19f 100644 --- a/crates/asap-physical-operators/tests/promql_values.rs +++ b/crates/asap-physical-operators/tests/promql_values.rs @@ -1,7 +1,7 @@ //! Compile, persist and rebind dynamic-label computation without deployment lowering. use asap_physical_operators::{ operators::Operator, - physical_planner::{promql_values::*, CompiledPhysicalDag, Source}, + physical_planner::{promql_values::*, CompiledPhysicalDAG, Source}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; @@ -21,15 +21,15 @@ fn row(labels: &[(&str, &str)], value: f64) -> Vec { Value::Float64(value), ] } -fn run(graph: CompiledPhysicalDag, rows: Vec>) -> Vec> { - run_inputs(graph, vec![Batch::try_new(vector_schema(), rows).unwrap()]).unwrap() +fn run(dag: CompiledPhysicalDAG, rows: Vec>) -> Vec> { + run_inputs(dag, vec![Batch::try_new(vector_schema(), rows).unwrap()]).unwrap() } fn run_inputs( - graph: CompiledPhysicalDag, + dag: CompiledPhysicalDAG, batches: Vec, ) -> Result>, asap_physical_operators::Error> { - let graph = serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) - .unwrap(); + let dag = + serde_json::from_slice::(&serde_json::to_vec(&dag).unwrap()).unwrap(); let sources = batches .into_iter() .enumerate() @@ -41,7 +41,7 @@ fn run_inputs( ) }) .collect::>(); - let bound = graph.instantiate(sources).unwrap(); + let bound = dag.instantiate(sources).unwrap(); let context = RunContext::new( Scope::Query { evaluation_time_ms: 0, @@ -51,7 +51,7 @@ fn run_inputs( ) .unwrap(); block_on(async { - let mut stream = bound.execute(graph.roots(), context).unwrap().remove(0); + let mut stream = bound.execute(dag.roots(), context).unwrap().remove(0); let mut rows = Vec::new(); while let Some(batch) = stream.next().await { rows.extend(batch?.rows().iter().cloned()); @@ -138,10 +138,10 @@ fn empty_vector_aggregation_stays_empty() { assert!(matches!(scalar[0][0],Value::Float64(v) if v.is_nan())); } -// One persisted temporal graph accepts different request windows and detects resets. +// One persisted temporal DAG accepts different request windows and detects resets. #[test] -fn temporal_graph_uses_bound_window_without_recompilation() { - let graph = compile_temporal(&AggIntent::Rate, false).unwrap(); +fn temporal_dag_uses_bound_window_without_recompilation() { + let dag = compile_temporal(&AggIntent::Rate, false).unwrap(); for start in [0, 60_000] { let labels = row(&[("__name__", "counter"), ("job", "api")], 0.)[0].clone(); let samples = [(0, 5.), (30_000, 1.), (60_000, 7.)]; @@ -158,7 +158,7 @@ fn temporal_graph_uses_bound_window_without_recompilation() { }) .collect(); let output = run_inputs( - graph.clone(), + dag.clone(), vec![Batch::try_new(matrix_schema(), rows).unwrap()], ) .unwrap(); @@ -181,20 +181,20 @@ fn temporal_graph_uses_bound_window_without_recompilation() { Value::Timestamp(2000), ], ]; - assert!(run_inputs(graph, vec![Batch::try_new(matrix_schema(), rows).unwrap()]).is_err()); + assert!(run_inputs(dag, vec![Batch::try_new(matrix_schema(), rows).unwrap()]).is_err()); } // The quantile is an ordinary scalar input, and bucket labels are native computation. #[test] fn histogram_quantile_keeps_each_label_group() { - let graph = compile_histogram_quantile().unwrap(); + let dag = compile_histogram_quantile().unwrap(); let buckets = vec![ row(&[("job", "api"), ("le", "1")], 2.), row(&[("job", "api"), ("le", "2")], 4.), row(&[("job", "api"), ("le", "+Inf")], 4.), ]; let output = run_inputs( - graph, + dag, vec![ Batch::try_new(scalar_schema(), vec![vec![Value::Float64(0.75)]]).unwrap(), Batch::try_new(vector_schema(), buckets).unwrap(), @@ -263,7 +263,7 @@ fn composed_ensemble_shares_a_producer_across_roots() { false, ) .unwrap(); - let graph = CompiledPhysicalDag::compose( + let dag = CompiledPhysicalDAG::compose( BTreeMap::from([(0, InputContract::bounded(vector_schema()))]), BTreeMap::from([ (10, (vec![0], aggregate)), @@ -273,9 +273,9 @@ fn composed_ensemble_shares_a_producer_across_roots() { vec![20, 30], ) .unwrap(); - let graph = serde_json::from_slice::(&serde_json::to_vec(&graph).unwrap()) - .unwrap(); - assert_eq!(graph.input_contracts().count(), 1); + let dag = + serde_json::from_slice::(&serde_json::to_vec(&dag).unwrap()).unwrap(); + assert_eq!(dag.input_contracts().count(), 1); let starts = std::rc::Rc::new(std::cell::Cell::new(0)); for _ in 0..2 { let input = Batch::try_new(vector_schema(), vec![row(&[("job", "api")], 3.)]).unwrap(); @@ -283,7 +283,7 @@ fn composed_ensemble_shares_a_producer_across_roots() { source: Operator::source(vector_schema(), vec![input]).unwrap(), starts: starts.clone(), }; - let bound = graph + let bound = dag .instantiate(BTreeMap::from([(0, Box::new(source) as Source<'_>)])) .unwrap(); let context = RunContext::new( @@ -300,7 +300,7 @@ fn composed_ensemble_shares_a_producer_across_roots() { let results = block_on(futures::future::join_all( bound - .execute(graph.roots(), context) + .execute(dag.roots(), context) .unwrap() .into_iter() .map(|mut stream| async move { @@ -315,9 +315,9 @@ fn composed_ensemble_shares_a_producer_across_roots() { #[test] fn compiled_constant_needs_no_deployment_source() { - let graph = compile_scalar(3.).unwrap(); - assert_eq!(graph.input_contracts().count(), 0); - let result = run_inputs(graph, vec![]).unwrap(); + let dag = compile_scalar(3.).unwrap(); + assert_eq!(dag.input_contracts().count(), 0); + let result = run_inputs(dag, vec![]).unwrap(); assert!(matches!(result[0][0], Value::Float64(3.))); } @@ -335,7 +335,7 @@ fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { (BinaryOpKind::Arithmetic(ArithmeticOpKind::Add), false), (BinaryOpKind::Compare(CompareOpKind::Gt), true), ] { - let graph = compile_binary( + let dag = compile_binary( &BinaryOperator { kind, vector_match: None, @@ -358,7 +358,7 @@ fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { let scalar = Batch::try_new(scalar_schema(), vec![vec![Value::Float64(1.)]]).unwrap(); let result = run_inputs( - graph, + dag, if left_scalar { vec![scalar, vector] } else { @@ -369,7 +369,7 @@ fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { } } } - let graph = compile_binary( + let dag = compile_binary( &BinaryOperator { kind: BinaryOpKind::Compare(CompareOpKind::Gt), vector_match: None, @@ -387,7 +387,7 @@ fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { ]; equal_rows( run_inputs( - graph, + dag, vec![ Batch::try_new(vector_schema(), rows.clone()).unwrap(), Batch::try_new(scalar_schema(), vec![vec![Value::Float64(1.)]]).unwrap(), @@ -398,7 +398,7 @@ fn scalar_broadcast_rejects_colliding_result_labels_after_recovery() { ); } -// Persisted exact readout graphs, rather than the storage adapter, merge panes, +// Persisted exact readout DAGs, rather than the storage adapter, merge panes, // finalize each population, and preserve the requested metric-name semantics. #[test] fn exact_state_readouts_recover_and_finalize_panes() { diff --git a/crates/asap-physical-operators/tests/raw_scan.rs b/crates/asap-physical-operators/tests/raw_scan.rs index 78e4c8c02..3e9f4cb10 100644 --- a/crates/asap-physical-operators/tests/raw_scan.rs +++ b/crates/asap-physical-operators/tests/raw_scan.rs @@ -53,15 +53,15 @@ fn fixture() -> (QueryExpr, Schema, Vec) { ]; (scan, output, batches) } -fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> PostAsapDag { - let node = |id, payload| PostAsapDagNode { +fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> PostAsapDAG { + let node = |id, payload| PostAsapDAGNode { id: PostAsapNodeId(id), payload, output_state: state, output_schema: (**schema).clone(), guarantee: None, }; - let edge = |producer, consumer| PostAsapDagEdge { + let edge = |producer, consumer| PostAsapDAGEdge { producer: PostAsapNodeId(producer), consumer: PostAsapNodeId(consumer), role: EdgeRole::Input, @@ -70,7 +70,7 @@ fn plan(scan: QueryExpr, schema: &Schema, state: ExecutionDataState) -> PostAsap grouping: GroupingEdgeCompatibility::NotApplicable, window: WindowEdgeCompatibility::NotApplicable, }; - PostAsapDag { + PostAsapDAG { nodes: vec![ node(0, PostAsapOperatorPayload::Fallback { expression: scan }), node( @@ -264,13 +264,13 @@ fn schema_drift_and_memory_limits_fail_the_scan() { batch: bad, })); let plan = plan(scan, &schema, ExecutionDataState::QUERY_ROWS); - let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + let physical_dag = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); block_on(async { - let mut s = graph.execute(&[0], context()).unwrap().remove(0); + let mut s = physical_dag.execute(&[0], context()).unwrap().remove(0); assert!(s.next().await.unwrap().is_err()); }); let sources = registry(Arc::new(MemorySource::new(schema, batches).unwrap())); - let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + let physical_dag = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); let ctx = RunContext::new( Scope::Query { evaluation_time_ms: 0, @@ -283,7 +283,7 @@ fn schema_drift_and_memory_limits_fail_the_scan() { ) .unwrap(); block_on(async { - let mut s = graph.execute(&[0], ctx.clone()).unwrap().remove(0); + let mut s = physical_dag.execute(&[0], ctx.clone()).unwrap().remove(0); assert!(s.next().await.unwrap().is_err()); }); assert_eq!(ctx.retained_bytes(), 0); @@ -318,9 +318,9 @@ fn empty_sources_and_three_valued_predicates() { ) .unwrap(); let plan = plan(scan.clone(), &schema, ExecutionDataState::QUERY_ROWS); - let graph = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); + let physical_dag = bind_with_data_sources(&plan, BTreeMap::new(), &[0], &sources).unwrap(); block_on(async { - let mut s = graph.execute(&[0], context()).unwrap().remove(0); + let mut s = physical_dag.execute(&[0], context()).unwrap().remove(0); let mut count = 0; while let Some(b) = s.next().await { count += b.unwrap().rows().len(); @@ -351,8 +351,8 @@ fn compile_without_readers_and_rebind_inputs() { 0, Box::new(Operator::source(schema.clone(), batches.clone()).unwrap()) as Source<'_>, )]); - let graph = compiled.instantiate(sources).unwrap(); - let mut outputs = graph.execute(compiled.roots(), context()).unwrap(); + let physical_dag = compiled.instantiate(sources).unwrap(); + let mut outputs = physical_dag.execute(compiled.roots(), context()).unwrap(); let result = block_on(outputs.remove(0).collect::>()); assert!(result.iter().all(Result::is_ok)); assert_eq!( diff --git a/crates/asap-physical-operators/tests/summary_projection.rs b/crates/asap-physical-operators/tests/summary_projection.rs index a61a596cb..63785a1b3 100644 --- a/crates/asap-physical-operators/tests/summary_projection.rs +++ b/crates/asap-physical-operators/tests/summary_projection.rs @@ -3,7 +3,7 @@ use asap_physical_operators::{ expressions::Expression, factory::create_planner_accumulator, operators::Operator, - physical_planner::{compile, CompiledPhysicalDag, InputContract, Source}, + physical_planner::{compile, CompiledPhysicalDAG, InputContract, Source}, runtime::{Limits, RunContext, Scope}, values::{Batch, Value}, }; @@ -44,16 +44,16 @@ fn post_asap_summary_projection_survives_recovery() { ], time_index: None, }; - let dag = PostAsapDag { + let dag = PostAsapDAG { nodes: vec![ - PostAsapDagNode { + PostAsapDAGNode { id: PostAsapNodeId(0), payload: PostAsapOperatorPayload::SummaryMerge, output_schema: (*schema).clone(), output_state: ExecutionDataState::INGESTION_SUMMARY, guarantee: None, }, - PostAsapDagNode { + PostAsapDAGNode { id: PostAsapNodeId(1), payload: PostAsapOperatorPayload::Value { operation: ValueOperation::Project { @@ -72,7 +72,7 @@ fn post_asap_summary_projection_survives_recovery() { guarantee: None, }, ], - edges: vec![PostAsapDagEdge { + edges: vec![PostAsapDAGEdge { producer: PostAsapNodeId(0), consumer: PostAsapNodeId(1), role: EdgeRole::Input, @@ -90,12 +90,12 @@ fn post_asap_summary_projection_survives_recovery() { ) .unwrap(); let encoded = serde_json::to_vec(&program).unwrap(); - let program = serde_json::from_slice::(&encoded).unwrap(); + let program = serde_json::from_slice::(&encoded).unwrap(); let mut forged: serde_json::Value = serde_json::from_slice(&encoded).unwrap(); forged["nodes"]["1"]["Operator"]["operator"]["output"]["fields"][1]["dtype"] = serde_json::json!({"Plain": "float64"}); assert!( - serde_json::from_slice::(&serde_json::to_vec(&forged).unwrap()) + serde_json::from_slice::(&serde_json::to_vec(&forged).unwrap()) .is_err() ); assert!(Operator::project( @@ -131,7 +131,7 @@ fn post_asap_summary_projection_survives_recovery() { ]], ) .unwrap(); - let graph = program + let physical_dag = program .instantiate(BTreeMap::from([( 0, Box::new(Operator::source(schema, vec![batch]).unwrap()) as Source<'_>, @@ -146,7 +146,7 @@ fn post_asap_summary_projection_survives_recovery() { ) .unwrap(); block_on(async { - let mut output = graph.execute(&[1], context).unwrap().remove(0); + let mut output = physical_dag.execute(&[1], context).unwrap().remove(0); let batch = output.next().await.unwrap().unwrap(); assert!(matches!(&batch.rows()[0][0], Value::Utf8(label) if label.as_ref() == "api")); let Value::Summary { diff --git a/crates/asap-physical-operators/tests/weighted_topk_binding.rs b/crates/asap-physical-operators/tests/weighted_topk_binding.rs index ef086b1bf..f5e875bab 100644 --- a/crates/asap-physical-operators/tests/weighted_topk_binding.rs +++ b/crates/asap-physical-operators/tests/weighted_topk_binding.rs @@ -159,13 +159,13 @@ fn assert_weighted_binding(evidence: &dyn AccuracyEvidenceProvider, algorithm: S &[dag.root.0 as u64], ) .unwrap(); - let graph = compiled + let physical_dag = compiled .instantiate(BTreeMap::from([(rate_id.0 as u64, source)])) .unwrap(); let context = RunContext::new(scope, Limits::default()).unwrap(); let output = block_on(async { let mut output = Vec::new(); - let mut stream = graph + let mut stream = physical_dag .execute(&[dag.root.0 as u64], context) .unwrap() .remove(0); @@ -395,7 +395,7 @@ fn check_direct_rate_topk(dynamic: bool) { .unwrap(); let bytes = serde_json::to_vec(&raw_compiled).unwrap(); let raw_compiled = serde_json::from_slice::< - asap_physical_operators::physical_planner::CompiledPhysicalDag, + asap_physical_operators::physical_planner::CompiledPhysicalDAG, >(&bytes) .unwrap(); // Each evaluation receives a complete raw window. A reset, a stopped @@ -468,13 +468,13 @@ fn check_direct_rate_topk(dynamic: bool) { let source = Box::new( Operator::source(raw_schema.clone(), vec![raw_batch.clone()]).unwrap(), ) as Source<'static>; - let graph = raw_compiled + let physical_dag = raw_compiled .instantiate(BTreeMap::from([(u64::from(raw.id.0), source)])) .unwrap(); let context = RunContext::new(scope, Limits::default()).unwrap(); let mut raw_scores = block_on(async { let mut scores = Vec::new(); - let mut stream = graph + let mut stream = physical_dag .execute(&[u64::from(dag.root.0)], context) .unwrap() .remove(0); @@ -584,13 +584,13 @@ fn check_direct_rate_topk(dynamic: bool) { let source = Box::new(Operator::source(schema.clone(), vec![batch.clone()]).unwrap()) as Source<'static>; - let graph = compiled + let physical_dag = compiled .instantiate(BTreeMap::from([(u64::from(input_id.0), source)])) .unwrap(); let context = RunContext::new(scope, Limits::default()).unwrap(); let mut scores = block_on(async { let mut scores = vec![]; - let mut stream = graph + let mut stream = physical_dag .execute(&[u64::from(dag.root.0)], context) .unwrap() .remove(0); @@ -707,7 +707,7 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { }) .collect(); let batch = Batch::try_new(schema.clone(), rows).unwrap(); - let graph = program + let physical_dag = program .instantiate(BTreeMap::from([( u64::from(raw.id.0), Box::new(Operator::source(schema.clone(), vec![batch]).unwrap()) as Source<'_>, @@ -722,7 +722,10 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { Limits::default(), ) .unwrap(); - let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let mut stream = physical_dag + .execute(program.roots(), context) + .unwrap() + .remove(0); let mut result = Vec::new(); while let Some(batch) = stream.next().await { let batch = batch.unwrap(); @@ -758,7 +761,7 @@ fn spatial_topk_exposes_signed_heap_candidate_over_complete_snapshot() { /// Deployment-side lifecycle choice: every summary state of `candidate` is /// continuously maintained, and the chosen lifecycles set execution timing. -fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDag { +fn continuously_maintained_dag(candidate: &Rc) -> PostAsapDAG { use asap_aware_mapping::{ cost_model::{Cost, CostModel}, enumerate_summary_maintenance_lifecycles, CostRate, Horizon, @@ -922,15 +925,15 @@ fn maintained_rate_heap_lifecycle_compiles_fixed_window_precompute() { ); // Execute the selected split across a state serialization boundary. // Each run builds fresh weights from that window's counters. - let execute = |plan: &asap_physical_operators::physical_planner::CompiledPhysicalDag, + let execute = |plan: &asap_physical_operators::physical_planner::CompiledPhysicalDAG, input: Batch, scope: Scope| { let id = plan.input_contracts().next().unwrap().0; let source = Box::new(Operator::source(input.schema().clone(), vec![input]).unwrap()) as Source<'static>; - let graph = plan.instantiate(BTreeMap::from([(id, source)])).unwrap(); + let physical_dag = plan.instantiate(BTreeMap::from([(id, source)])).unwrap(); block_on(async { - let mut stream = graph + let mut stream = physical_dag .execute( plan.roots(), RunContext::new(scope, Limits::default()).unwrap(), diff --git a/crates/devtools/src/bin/analyze_corpora.rs b/crates/devtools/src/bin/analyze_corpora.rs index 503fb1f04..2c4bacefb 100644 --- a/crates/devtools/src/bin/analyze_corpora.rs +++ b/crates/devtools/src/bin/analyze_corpora.rs @@ -529,7 +529,7 @@ async fn run_sql_corpora(out_dir: PathBuf) { serde_json::to_vec_pretty(&summary).unwrap(), ) .expect("failed to write summary.json"); - let notes = "- **Manual review result:** the reviewed SQL trees preserve `COUNT(*)` versus `COUNT(column)`, `COUNT(DISTINCT)`/`uniqExact`, `HAVING`, CTE-derived projections, `LAG`/`lagInFrame`, grouping keys, and explicit window frames. No concrete semantic collapse was found in this pass.\n- **Apparently intentional omission:** ClickHouse `FORMAT Null` is absent from the IR; it is an output/transport directive rather than query semantics.\n- **Failure boundaries:** unsupported ClickHouse functions and unsupported grammar are retained in the per-query error files rather than being converted into partial IR.\n"; + let notes = "- **Manual review result:** the reviewed SQL DAGs preserve `COUNT(*)` versus `COUNT(column)`, `COUNT(DISTINCT)`/`uniqExact`, `HAVING`, CTE-derived projections, `LAG`/`lagInFrame`, grouping keys, and explicit window frames. No concrete semantic collapse was found in this pass.\n- **Apparently intentional omission:** ClickHouse `FORMAT Null` is absent from the IR; it is an output/transport directive rather than query semantics.\n- **Failure boundaries:** unsupported ClickHouse functions and unsupported grammar are retained in the per-query error files rather than being converted into partial IR.\n"; std::fs::write( out_dir.join("anomalies.md"), anomaly_report(&all, "SQL", notes), diff --git a/crates/devtools/src/bin/dag_export.rs b/crates/devtools/src/bin/dag_export.rs index 3f7cd8850..9c9829e78 100644 --- a/crates/devtools/src/bin/dag_export.rs +++ b/crates/devtools/src/bin/dag_export.rs @@ -3,7 +3,7 @@ // --promql "topk(5, rate(http_requests_total[5m]))" --name q2 // // Lowers each given SQL/PromQL query to pre-ASAP IR and prints a single -// `asap_types::dag_export::WorkloadGraph` as JSON on stdout — the input format +// `asap_types::dag_export::WorkloadDAG` as JSON on stdout — the input format // for `tools/dag-viewer` (issue #133). Redirect to a file and load it there: // cargo run -p asap-lower --bin dag_export -- --sql "..." --name q1 > /tmp/dag.json // @@ -31,23 +31,23 @@ // candidate per group feeds two additive outputs: // // - one `asap_types::dag_export::TargetReplacement` per group on whichever -// query's `NamedGraph.replacements` contains that target node (matched -// by `DagNode::hash` + structural equality, the same collision-safe +// query's `NamedDAG.replacements` contains that target node (matched +// by `DAGNode::hash` + structural equality, the same collision-safe // pattern `annotate_with_explanations` below already uses for notes) — // a small, self-contained "before -> after" pair per replacement site; -// - one merged `NamedGraph.post_graph`: a single flattened graph per +// - one merged `NamedDAG.post_dag`: a single flattened DAG per // query with every winning candidate spliced directly into the query's // own pre-ASAP shape in place, built via // `asap_types::dag_export::export_post_asap`. // // Together these surface every one of the four concrete replacement kinds: // the sketch family `SketchAlgorithmStrategy`/`HydraGroupingStrategy` bound, -// the CSE share/recompute choice `SharedSubtreeStrategy` found, the +// the CSE share/recompute choice `SharedSubDAGStrategy` found, the // workload-aware roll-up `RollupStrategy` derived, and the `avg -> // sum/count` rewrite `AvgToSumOverCountStrategy` proposes. Without // `--post-asap`, every existing invocation of this binary produces -// byte-identical output to before (`NamedGraph.replacements` is empty and -// `post_graph` is `None`, both skipped from the JSON entirely in that +// byte-identical output to before (`NamedDAG.replacements` is empty and +// `post_dag` is `None`, both skipped from the JSON entirely in that // case). E.g.: // cargo run -p asap-lower --bin dag_export -- \ // --post-asap --default-cost --epsilon 0.01 \ @@ -77,7 +77,7 @@ use std::rc::Rc; use std::time::Instant; use asap_aware_mapping::analytical_cost::{ - cache_hit_ratios, AnalyticalCostError, EvidenceBackedPhysicalDag as PhysicalDag, + cache_hit_ratios, AnalyticalCostError, EvidenceBackedPhysicalDAG as PhysicalDAG, PhysicalNodeEvidence, ResourceCalibration, ANALYTICAL_COST_MODEL_VERSION, }; use asap_aware_mapping::cost_model::DefaultCostModel; @@ -94,8 +94,8 @@ use asap_aware_mapping::replacement::{ use asap_aware_mapping::{AccuracyEvidenceProvider, PropagationStats}; use asap_types::cost::{BaselineRef, CostAnnotation, CostInput, CostSource, CostUnit}; use asap_types::dag_export::{ - self, DagDecision, DagGraph, DagNote, NamedGraph, PostAsapSubstitution, TargetRejection, - TargetReplacement, TargetReplacementAfter, WorkloadGraph, + self, DAGDecision, DAGNote, ExportDAG, NamedDAG, PostAsapSubstitution, TargetRejection, + TargetReplacement, TargetReplacementAfter, WorkloadDAG, }; use asap_types::post_asap::SummaryExpr; use asap_types::post_asap::SummaryNode; @@ -209,7 +209,7 @@ enum CandidatePhysicalEvidence { Summary { plan: serde_json::Value, query_nodes: Vec, - physical_dag: PhysicalDag, + physical_dag: PhysicalDAG, }, } @@ -239,7 +239,7 @@ impl CandidatePhysicalEvidence { actual.is_ok_and(|actual| plan_values_match(&actual, self.plan())) } - fn summary_dag(&self) -> Option<&PhysicalDag> { + fn summary_dag(&self) -> Option<&PhysicalDAG> { match self { Self::Summary { physical_dag, .. } => Some(physical_dag), Self::Rewrite { .. } => None, @@ -249,7 +249,7 @@ impl CandidatePhysicalEvidence { /// Compare complete exported-plan identity while tolerating the one-ULP /// decimal round trip that `serde_json::Value` can introduce for derived -/// floating-point guarantees. Integer configuration and graph identity stay +/// floating-point guarantees. Integer configuration and DAG identity stay /// exact; no guarantee field is dropped or otherwise normalized away. fn plan_values_match(actual: &serde_json::Value, expected: &serde_json::Value) -> bool { plan_values_match_inner(actual, expected, false) @@ -358,7 +358,7 @@ impl PlannerPhysicalPlanProvider for ExportPhysicalProvider<'_> { )) })?; if matches.next().is_some() { - return Err(AnalyticalCostError::InvalidPhysicalDag( + return Err(AnalyticalCostError::InvalidPhysicalDAG( "duplicate query-node evidence key", )); } @@ -371,7 +371,7 @@ impl PlannerPhysicalPlanProvider for ExportPhysicalProvider<'_> { snapshot: &PhysicalEvidenceSnapshot, _summary: &Rc, _target: &asap_aware_mapping::replacement::TargetSubDAG<'_>, - ) -> Result { + ) -> Result { if snapshot.scope != self.target.scope.resolve()? { return Err(AnalyticalCostError::ComparisonScopeMismatch( "planner evidence snapshot", @@ -380,7 +380,7 @@ impl PlannerPhysicalPlanProvider for ExportPhysicalProvider<'_> { self.candidate .summary_dag() .cloned() - .ok_or(AnalyticalCostError::InvalidPhysicalDag( + .ok_or(AnalyticalCostError::InvalidPhysicalDAG( "summary candidate is missing its physical DAG", )) } @@ -682,7 +682,7 @@ impl CostModel for ExportPlannerCostModel<'_> { /// Baseline/selected/benefit [`CostAnnotation`]s for one [`Winner`] — issue /// #286's "replacement-region baseline cost, selected cost, and benefit" /// granularity item, reused verbatim for [`TargetReplacement`] and for the -/// [`DagDecision`] carried by every node the winning candidate produced or +/// [`DAGDecision`] carried by every node the winning candidate produced or /// carried. /// /// `dag_export` has no deployment-owned physical evidence provider. It must @@ -960,23 +960,23 @@ fn parse_args_from(argv: impl Iterator) -> ParsedArgs { } } -/// Attach workload-wide replacement explanations to their exact graph nodes. +/// Attach workload-wide replacement explanations to their exact DAG nodes. /// `node_hash` is only a narrowing filter; `source_expr == Some(target)` is /// the collision-safe identity check (`source_expr` is `None` only for a -/// post-ASAP-originated node inside a `--post-asap` `post_graph`, which this +/// post-ASAP-originated node inside a `--post-asap` `post_dag`, which this /// function is never called on — every node it sees, from an ordinary /// [`dag_export::export`], carries `Some`). fn annotate_with_explanations( - graph: &mut DagGraph, + dag: &mut ExportDAG, explanations: &[asap_aware_mapping::ReplacementExplanation], matched: &mut [bool], ) { for (i, explanation) in explanations.iter().enumerate() { - for node in graph.nodes.iter_mut() { + for node in dag.nodes.iter_mut() { if node.hash == Some(explanation.node_hash) && node.source_expr.as_ref() == Some(explanation.target.as_ref()) { - node.notes.push(DagNote { + node.notes.push(DAGNote { kind: format!("{:?}", explanation.kind), reason: explanation.reason.clone(), }); @@ -988,7 +988,7 @@ fn annotate_with_explanations( /// One `TargetSubDAGCandidates`'s best-ranked candidate, kept alongside its own `target` /// — the unit both [`PostAsapResults::replacements`] and -/// [`PostAsapResults::post_graphs`] are built from, so the two outputs can +/// [`PostAsapResults::post_dags`] are built from, so the two outputs can /// never disagree about which candidate won for a given target. #[allow(dead_code)] struct Winner<'a> { @@ -1012,15 +1012,15 @@ fn decision_rationale(winner: &Winner<'_>) -> String { "Composes compatible nested aggregates using their declared algebraic intent while preserving the output schema." .to_string() } - "SharedSubtreeStrategy" => match winner.candidate.provenance { + "SharedSubDAGStrategy" => match winner.candidate.provenance { asap_aware_mapping::replacement::ReplacementProvenance::CseShare => { - "Builds the repeated subtree once and shares it across consumers.".to_string() + "Builds the repeated sub-DAG once and shares it across consumers.".to_string() } asap_aware_mapping::replacement::ReplacementProvenance::CseRecompute => { - "Recomputes the subtree per consumer because that has the lower estimated cost." + "Recomputes the sub-DAG per consumer because that has the lower estimated cost." .to_string() } - _ => "Chooses the lowest-cost handling of the repeated subtree.".to_string(), + _ => "Chooses the lowest-cost handling of the repeated sub-DAG.".to_string(), }, "RollupStrategy" => { "Answers this aggregate from a compatible finer-grained aggregate.".to_string() @@ -1067,15 +1067,15 @@ fn lookup_winner( } /// One `(decision.id, baseline_cost, selected_cost)` triple per *distinct* -/// [`DagDecision`] carried anywhere in `graph` — collapsing every node that +/// [`DAGDecision`] carried anywhere in `dag` — collapsing every node that /// shares one `decision.id` (a replacement region can span many nodes, all /// carrying an identical clone of the same decision) down to a single /// entry, so a caller summing these never counts one decision's cost once /// per node it happens to touch. -fn decision_cost_entries(graph: &DagGraph) -> Vec<(u32, CostAnnotation, CostAnnotation)> { +fn decision_cost_entries(dag: &ExportDAG) -> Vec<(u32, CostAnnotation, CostAnnotation)> { let mut seen = std::collections::HashSet::new(); let mut entries = Vec::new(); - for node in &graph.nodes { + for node in &dag.nodes { let Some(decision) = &node.decision else { continue; }; @@ -1135,44 +1135,44 @@ fn target_replacement( /// usage doc for what each is for. struct PostAsapResults { /// One `(query_name, TargetReplacement)` pair per discovered replacement - /// site whose target node is found in that query's own exported graph. A + /// site whose target node is found in that query's own exported DAG. A /// target can in principle be reachable from more than one query's root - /// after CSE (a shared subtree), in which case it yields one pair per + /// after CSE (a shared sub-DAG), in which case it yields one pair per /// matching query, each with that query's own `target_pre_id`. replacements: Vec<(String, TargetReplacement)>, - /// One merged, whole-query [`DagGraph`] per query, built via + /// One merged, whole-query [`ExportDAG`] per query, built via /// [`dag_export::export_post_asap`] — every winning candidate spliced /// directly into that query's own pre-ASAP shape in place. - post_graphs: Vec<(String, DagGraph)>, + post_dags: Vec<(String, ExportDAG)>, /// One `(query_name, TargetRejection)` per accuracy-illegal candidate /// the search refused (`TargetSubDAGCandidates::rejected`, issue #172) whose target - /// node is found in that query's own exported graph. + /// node is found in that query's own exported DAG. rejections: Vec<(String, TargetRejection)>, } fn raw_only_post_asap_results() -> PostAsapResults { PostAsapResults { replacements: Vec::new(), - post_graphs: Vec::new(), + post_dags: Vec::new(), rejections: Vec::new(), } } /// Assign collision-free, explicit identities to structurally equal nodes -/// across a set of exported query graphs. The full canonical subtree string +/// across a set of exported query DAGs. The full canonical sub-DAG string /// is the equality key; the compact integer is what JSON consumers receive. /// Consequently the viewer never needs to guess identity from labels, /// hashes, or a client-side node signature. -fn assign_workload_node_ids(graphs: &mut [&mut DagGraph]) { - fn key_for(id: u32, graph: &DagGraph, memo: &mut HashMap) -> String { +fn assign_workload_node_ids(dags: &mut [&mut ExportDAG]) { + fn key_for(id: u32, dag: &ExportDAG, memo: &mut HashMap) -> String { if let Some(key) = memo.get(&id) { return key.clone(); } - let node = &graph.nodes[id as usize]; + let node = &dag.nodes[id as usize]; let child_keys: Vec<_> = node .children .iter() - .map(|child| key_for(*child, graph, memo)) + .map(|child| key_for(*child, dag, memo)) .collect(); let key = serde_json::to_string(&(node.kind, &node.detail, &node.schema, child_keys)) .expect("exported DAG node content is serializable"); @@ -1182,14 +1182,14 @@ fn assign_workload_node_ids(graphs: &mut [&mut DagGraph]) { let mut ids = HashMap::::new(); let mut next_id = 0_u32; - for graph in graphs.iter_mut() { + for dag in dags.iter_mut() { let mut memo = HashMap::new(); - let keys: Vec<_> = graph + let keys: Vec<_> = dag .nodes .iter() - .map(|node| key_for(node.id, graph, &mut memo)) + .map(|node| key_for(node.id, dag, &mut memo)) .collect(); - for (node, key) in graph.nodes.iter_mut().zip(keys) { + for (node, key) in dag.nodes.iter_mut().zip(keys) { let id = *ids.entry(key).or_insert_with(|| { let id = next_id; next_id += 1; @@ -1206,7 +1206,7 @@ fn assign_workload_node_ids(graphs: &mut [&mut DagGraph]) { /// needed) over every lowered query, rank each discovered `TargetSubDAGCandidates` via /// `CandidateLogicalASAPDAGs::global_selection`, and build both `--post-asap` outputs from the /// exact same set of winning candidates (see [`Winner`]), so the flat -/// `replacements` list and the merged `post_graph` can never disagree about +/// `replacements` list and the merged `post_dag` can never disagree about /// which candidate won for a given target. #[allow(dead_code)] fn run_post_asap_with_progress( @@ -1298,7 +1298,7 @@ fn run_post_asap_with_progress( ); } - // ---- Merged, whole-query `post_graph`, one per query --------------- + // ---- Merged, whole-query `post_dag`, one per query --------------- // // Built *before* the flat `replacements` pass below, not after: a // winner whose target only exists inside another winner's own @@ -1306,10 +1306,10 @@ fn run_post_asap_with_progress( // `AvgToSumOverCountStrategy`'s rewrite exposes, which // `default_strategies()` — #282 — now discovers and independently // sketch-ranks in the same search pass) can never appear in any query's - // *original*, pre-rewrite `graph` — there's nothing wrong with that - // winner, it's just nested. `post_graph` is where it's expected to + // *original*, pre-rewrite `dag` — there's nothing wrong with that + // winner, it's just nested. `post_dag` is where it's expected to // surface instead (`export_post_asap`'s recursive `find_winner` - // threading walks straight through a rewritten subtree and re-checks + // threading walks straight through a rewritten sub-DAG and re-checks // every node inside it too), so the flat-`replacements` pass below // checks there before deciding a miss is a real anomaly worth a // warning. @@ -1317,15 +1317,15 @@ fn run_post_asap_with_progress( eprintln!("[4/4] Post-ASAP DAG generation is running…"); } let post_started = Instant::now(); - let mut post_graph_cache = HashCache::new(); + let mut post_dag_cache = HashCache::new(); let mut find_winner = |expr: &QueryExpr| -> Option { - let i = lookup_winner(&by_hash, &winners, &mut post_graph_cache, expr)?; + let i = lookup_winner(&by_hash, &winners, &mut post_dag_cache, expr)?; let winner = &winners[i]; let (baseline_cost, selected_cost, benefit) = winner.costs.clone(); // Derived from `selected_cost`; see `target_replacement`'s identical // derivation. let cost = selected_cost.value.unwrap_or(f64::NAN); - let decision = DagDecision { + let decision = DAGDecision { id: i as u32, strategy: winner.candidate.strategy.to_string(), rationale: decision_rationale(winner), @@ -1350,7 +1350,7 @@ fn run_post_asap_with_progress( } }) }; - let post_graphs: Vec<(String, DagGraph)> = lowered_queries + let post_dags: Vec<(String, ExportDAG)> = lowered_queries .iter() .map(|(name, _, qe)| { ( @@ -1362,21 +1362,21 @@ fn run_post_asap_with_progress( // ---- Flat per-target `replacements`, one list per query ----------- // - // Independently re-export every query's own graph for matching — a + // Independently re-export every query's own DAG for matching — a // fresh `export` per query, not reused from `main`'s own already-built - // `NamedGraph`s, so this function stays self-contained and callable on + // `NamedDAG`s, so this function stays self-contained and callable on // its own (see this file's `#[cfg(test)]` module). Deliberately anchored - // to the *original* `graph` only (never `post_graph`) — `target_pre_id` - // is documented as an id into `NamedGraph.graph.nodes`, so a nested + // to the *original* `dag` only (never `post_dag`) — `target_pre_id` + // is documented as an id into `NamedDAG.DAG.nodes`, so a nested // secondary target (see above) never gets a flat entry of its own here: // it's already visible, in place, inside its parent's own `after` - // subtree and inside `post_graph` as a whole. + // sub-DAG and inside `post_dag` as a whole. let mut lookup_cache = HashCache::new(); let mut replacements = Vec::new(); let mut rejections = Vec::new(); let mut matched = vec![false; winners.len()]; // Groups with accuracy-refused candidates (issue #172): matched to a - // query's graph nodes the same hash-then-structural-equality way. + // query's DAG nodes the same hash-then-structural-equality way. let rejected_groups: Vec<_> = space .target_subdag_candidates() .filter(|group| !group.rejected.is_empty()) @@ -1387,8 +1387,8 @@ fn run_post_asap_with_progress( rejected_by_hash.entry(hash).or_default().push(i); } for (name, _, qe) in lowered_queries { - let graph = dag_export::export(qe); - for node in &graph.nodes { + let dag = dag_export::export(qe); + for node in &dag.nodes { let Some(source_expr) = node.source_expr.as_ref() else { continue; // never true for a plain `export` — defensive only. }; @@ -1425,12 +1425,12 @@ fn run_post_asap_with_progress( // similarly `RollupStrategy`'s) can expose a brand-new `sum`/`count` // descendant that the *same* search pass then independently discovers // and ranks — a real winner, but one with no node anywhere in any - // query's original, pre-rewrite `graph` to attach a flat entry to - // (`target_pre_id` is documented as an id into `graph.nodes` + // query's original, pre-rewrite `dag` to attach a flat entry to + // (`target_pre_id` is documented as an id into `DAG.nodes` // specifically). This isn't a data loss: `export_post_asap` still - // splices that winner in, in place, inside `post_graph` — see this - // function's own construction of `post_graphs` above, which walks - // straight through a rewritten subtree and resolves every nested + // splices that winner in, in place, inside `post_dag` — see this + // function's own construction of `post_dags` above, which walks + // straight through a rewritten sub-DAG and resolves every nested // winner too, recursively. So an unmatched winner here is expected, // not necessarily a bug, whenever it's downstream of some other // winner's own `Replacement::Rewrite` — logged as an FYI rather than a @@ -1438,15 +1438,15 @@ fn run_post_asap_with_progress( // reimplementing `search`'s own private descendant-discovery walk // (`discover_new_descendant_targets` in `asap_aware_mapping::replacement`, // not exposed) a second time here just to double-check something - // `post_graph`'s own construction already handled correctly. + // `post_dag`'s own construction already handled correctly. for (winner, matched) in winners.iter().zip(&matched) { if !matched { let strategy = winner.candidate.strategy; eprintln!( "dag_export: post-asap replacement ({strategy}) has no node in any query's \ - original graph — expected for a winner exposed only inside another winner's \ + original DAG — expected for a winner exposed only inside another winner's \ own rewrite output (e.g. a sum/count descendant of an avg rewrite); still \ - present in that query's post_graph" + present in that query's post_dag" ); } } @@ -1459,7 +1459,7 @@ fn run_post_asap_with_progress( PostAsapResults { replacements, - post_graphs, + post_dags, rejections, } } @@ -1536,14 +1536,14 @@ async fn main() { let mut matched = vec![false; explanations.len()]; let mut queries = Vec::new(); for (name, source, qe) in &lowered_queries { - let mut graph = dag_export::export(qe); - annotate_with_explanations(&mut graph, &explanations, &mut matched); - queries.push(NamedGraph { + let mut dag = dag_export::export(qe); + annotate_with_explanations(&mut dag, &explanations, &mut matched); + queries.push(NamedDAG { name: name.clone(), source: Some(source.clone()), - graph, + dag, replacements: Vec::new(), - post_graph: None, + post_dag: None, workload_cost: None, rejections: Vec::new(), }); @@ -1551,7 +1551,7 @@ async fn main() { for (explanation, matched) in explanations.iter().zip(matched) { if !matched { eprintln!( - "dag_export: explanation at {} ({:?}, node_hash={}) matched no DagNode", + "dag_export: explanation at {} ({:?}, node_hash={}) matched no DAGNode", explanation.location, explanation.kind, explanation.node_hash ); } @@ -1606,9 +1606,9 @@ async fn main() { named.replacements.push(replacement); } } - for (query_name, post_graph) in results.post_graphs { + for (query_name, post_dag) in results.post_dags { if let Some(named) = queries.iter_mut().find(|q| q.name == query_name) { - named.post_graph = Some(post_graph); + named.post_dag = Some(post_dag); } } for (query_name, rejection) in results.rejections { @@ -1619,15 +1619,15 @@ async fn main() { } { - let mut pre_graphs: Vec<_> = queries.iter_mut().map(|query| &mut query.graph).collect(); - assign_workload_node_ids(&mut pre_graphs); + let mut pre_dags: Vec<_> = queries.iter_mut().map(|query| &mut query.dag).collect(); + assign_workload_node_ids(&mut pre_dags); } { - let mut post_graphs: Vec<_> = queries + let mut post_dags: Vec<_> = queries .iter_mut() - .filter_map(|query| query.post_graph.as_mut()) + .filter_map(|query| query.post_dag.as_mut()) .collect(); - assign_workload_node_ids(&mut post_graphs); + assign_workload_node_ids(&mut post_dags); } // Whole selected-workload cost/benefit (issue #286) — per query, and @@ -1641,10 +1641,10 @@ async fn main() { // needed. let mut workload_entries = Vec::new(); for query in &mut queries { - let Some(post_graph) = &query.post_graph else { + let Some(post_dag) = &query.post_dag else { continue; }; - let entries = decision_cost_entries(post_graph); + let entries = decision_cost_entries(post_dag); if entries.is_empty() { continue; } @@ -1679,7 +1679,7 @@ async fn main() { } }; - let workload = WorkloadGraph { + let workload = WorkloadDAG { queries, workload_cost, }; @@ -1697,7 +1697,7 @@ mod tests { use super::*; use asap_aware_mapping::analytical_cost::{ - ExecutionMultiplicity, PhysicalDagNode, PhysicalOperator, + ExecutionMultiplicity, PhysicalDAGNode, PhysicalOperator, }; use asap_aware_mapping::physical_operator_statistics::{ EdgeStatistics, OperatorStatistics, SourceCoverage, UnaryEdgeStatistics, @@ -1733,7 +1733,7 @@ mod tests { query: &QueryExpr, candidate: &ReplacementSubDAG, document: &PlannerCostDocument, - ) -> PhysicalDag { + ) -> PhysicalDAG { let model = ExportPlannerCostModel { document }; let root = Rc::new(query.clone()); let target = asap_aware_mapping::replacement::TargetSubDAG::new(&root); @@ -2246,7 +2246,7 @@ mod tests { entries.into_inner() } - fn cheap_candidate_dag() -> PhysicalDag { + fn cheap_candidate_dag() -> PhysicalDAG { let coverage = test_scope().sources[0].clone(); let statistics = OperatorStatistics::Scan { edges: UnaryEdgeStatistics { @@ -2256,8 +2256,8 @@ mod tests { }, source_read_bytes: 64, }; - PhysicalDag { - nodes: vec![PhysicalDagNode { + PhysicalDAG { + nodes: vec![PhysicalDAGNode { id: "summary-read".into(), operator: PhysicalOperator::Scan, children: vec![], @@ -2698,7 +2698,7 @@ mod tests { fn missing_physical_evidence_keeps_the_export_raw_only() { let results = raw_only_post_asap_results(); assert!(results.replacements.is_empty()); - assert!(results.post_graphs.is_empty()); + assert!(results.post_dags.is_empty()); } fn argv(args: &[&str]) -> impl Iterator { @@ -2850,12 +2850,12 @@ mod tests { .any(|e| { e.kind == asap_aware_mapping::ExplanationKind::CommonSubexpressionReuse })); let mut matched = vec![false; explanations.len()]; - let mut graph_a = dag_export::export(&a); - let mut graph_b = dag_export::export(&b); - annotate_with_explanations(&mut graph_a, &explanations, &mut matched); - annotate_with_explanations(&mut graph_b, &explanations, &mut matched); - assert!(graph_a.nodes.iter().any(|n| !n.notes.is_empty())); - assert!(graph_b.nodes.iter().any(|n| !n.notes.is_empty())); + let mut dag_a = dag_export::export(&a); + let mut dag_b = dag_export::export(&b); + annotate_with_explanations(&mut dag_a, &explanations, &mut matched); + annotate_with_explanations(&mut dag_b, &explanations, &mut matched); + assert!(dag_a.nodes.iter().any(|n| !n.notes.is_empty())); + assert!(dag_b.nodes.iter().any(|n| !n.notes.is_empty())); } #[test] @@ -2873,13 +2873,13 @@ mod tests { let unrelated = lower_promql("sum(rate(other_metric[5m]))", AccuracyTarget::Epsilon(0.01)).unwrap(); - let mut graph = dag_export::export(&unrelated); - for node in &mut graph.nodes { + let mut dag = dag_export::export(&unrelated); + for node in &mut dag.nodes { node.hash = Some(explanation.node_hash); } let mut matched = vec![false; explanations.len()]; - annotate_with_explanations(&mut graph, &explanations, &mut matched); - assert!(graph.nodes.iter().all(|n| n.notes.is_empty())); + annotate_with_explanations(&mut dag, &explanations, &mut matched); + assert!(dag.nodes.iter().all(|n| n.notes.is_empty())); } /// The `--post-asap` code path, exercised directly (not through the CLI): @@ -2888,7 +2888,7 @@ mod tests { /// `after: TargetReplacementAfter::Summary(..)` (the bound sketch) and /// at least one with `after: TargetReplacementAfter::Rewrite(..)` (the /// `avg -> sum/count` rewrite) — see [`run_post_asap`]. Also checks that - /// a non-empty `post_graph` comes back for every query, since that's the + /// a non-empty `post_dag` comes back for every query, since that's the /// other `--post-asap` output `main` wires up. #[tokio::test] async fn post_asap_run_produces_both_summary_and_rewrite_replacements() { @@ -2957,18 +2957,18 @@ mod tests { "AvgToSumOverCountStrategy" | "SemanticEquivalentRewriteStrategy" ))); - assert_eq!(results.post_graphs.len(), 2, "one post_graph per query"); - for (name, graph) in &results.post_graphs { + assert_eq!(results.post_dags.len(), 2, "one post_dag per query"); + for (name, dag) in &results.post_dags { assert!( - !graph.nodes.is_empty(), - "post_graph for {name:?} must not be empty" + !dag.nodes.is_empty(), + "post_dag for {name:?} must not be empty" ); } let avg_post = &results - .post_graphs + .post_dags .iter() .find(|(name, _)| name == "avg") - .expect("avg post graph") + .expect("avg post DAG") .1; assert_eq!( avg_post @@ -2980,13 +2980,13 @@ mod tests { "AVG's SUM and COUNT branches must retain their shared input as one DAG node" ); let decisions: Vec<_> = results - .post_graphs + .post_dags .iter() - .flat_map(|(_, graph)| graph.nodes.iter().filter_map(|node| node.decision.as_ref())) + .flat_map(|(_, dag)| dag.nodes.iter().filter_map(|node| node.decision.as_ref())) .collect(); assert!( !decisions.is_empty(), - "post_graph nodes must carry explicit strategy metadata" + "post_dag nodes must carry explicit strategy metadata" ); for decision in decisions { assert!(!decision.strategy.is_empty()); @@ -3021,21 +3021,17 @@ mod tests { ), ]; let mut results = run_post_asap(&lowered); - let mut graph_refs: Vec<_> = results - .post_graphs - .iter_mut() - .map(|(_, graph)| graph) - .collect(); - assign_workload_node_ids(&mut graph_refs); + let mut dag_refs: Vec<_> = results.post_dags.iter_mut().map(|(_, dag)| dag).collect(); + assign_workload_node_ids(&mut dag_refs); let q3 = &results - .post_graphs + .post_dags .iter() .find(|(name, _)| name == "q3") .unwrap() .1; let q4 = &results - .post_graphs + .post_dags .iter() .find(|(name, _)| name == "q4") .unwrap() @@ -3068,26 +3064,26 @@ mod tests { ) .await .unwrap(); - let mut q1_graph = dag_export::export(&q1); - let mut q6_graph = dag_export::export(&q6); - assign_workload_node_ids(&mut [&mut q1_graph, &mut q6_graph]); - let q1_scan = q1_graph + let mut q1_dag = dag_export::export(&q1); + let mut q6_dag = dag_export::export(&q6); + assign_workload_node_ids(&mut [&mut q1_dag, &mut q6_dag]); + let q1_scan = q1_dag .nodes .iter() .find(|node| node.label == "Scan(metrics)") .unwrap(); - let q6_scan = q6_graph + let q6_scan = q6_dag .nodes .iter() .find(|node| node.label == "Scan(metrics)") .unwrap(); assert_eq!(q1_scan.workload_node_id, q6_scan.workload_node_id); - assert!(q6_graph.nodes.iter().any(|node| node.kind == "Join")); + assert!(q6_dag.nodes.iter().any(|node| node.kind == "Join")); } /// The `--default-cost` contract, end to end on the code path `main` /// takes for it (`DefaultCostModel` ranking, `export_model: None`): the - /// structure must be real — replacements found, a merged `post_graph` + /// structure must be real — replacements found, a merged `post_dag` /// per query — while every cost stays `Unavailable` with no value, so /// the viewer shows "Not estimated" and the structural ranking number /// never escapes as if it were a measured cost. @@ -3114,9 +3110,9 @@ mod tests { "the search itself must still run under --default-cost" ); assert!(results - .post_graphs + .post_dags .iter() - .all(|(_, graph)| !graph.nodes.is_empty())); + .all(|(_, dag)| !dag.nodes.is_empty())); for (_, replacement) in &results.replacements { for annotation in [ @@ -3139,8 +3135,8 @@ mod tests { // Absent a value, the per-query aggregation `main` runs degrades to // an unavailable summary rather than a number or an error. - for (_, graph) in &results.post_graphs { - for (_, baseline, selected) in decision_cost_entries(graph) { + for (_, dag) in &results.post_dags { + for (_, baseline, selected) in decision_cost_entries(dag) { assert_eq!(baseline.source, CostSource::Unavailable); assert_eq!(selected.source, CostSource::Unavailable); } @@ -3195,7 +3191,7 @@ mod tests { .map(|(name, r)| (name.clone(), r.strategy.clone())) .collect::>() ); - assert_eq!(results.post_graphs.len(), 1); - assert!(!results.post_graphs[0].1.nodes.is_empty()); + assert_eq!(results.post_dags.len(), 1); + assert!(!results.post_dags[0].1.nodes.is_empty()); } } diff --git a/crates/devtools/src/bin/sketch_coverage.rs b/crates/devtools/src/bin/sketch_coverage.rs index e9f0ccb10..58fa25348 100644 --- a/crates/devtools/src/bin/sketch_coverage.rs +++ b/crates/devtools/src/bin/sketch_coverage.rs @@ -9,7 +9,7 @@ // - a `SketchApproximation` candidate (a genuine sketch alternative was // found for at least one aggregate in the query — the KLL-vs-DDSketch // kind of degree of freedom), and/or -// - a `CommonSubexpressionReuse` candidate (the query shares a subtree, +// - a `CommonSubexpressionReuse` candidate (the query shares a sub-DAG, // inside itself or with another query in the same corpus, that a // build-once-and-share candidate was found for). // diff --git a/crates/devtools/src/bin/variant_coverage.rs b/crates/devtools/src/bin/variant_coverage.rs index a96042a2e..290d5584d 100644 --- a/crates/devtools/src/bin/variant_coverage.rs +++ b/crates/devtools/src/bin/variant_coverage.rs @@ -1,7 +1,7 @@ // cargo run -p asap-lower --bin variant_coverage // // Lowers every query in every corpus we have (PromQL + SQL), walks the -// resulting QueryExpr trees, and reports which enum variants show up — per +// resulting QueryExpr DAGs, and reports which enum variants show up — per // corpus, then rolled up globally. Used to find the minimal QueryExpr node set. use asap_devtools::lower_promql_with_data_ingestion_interval; diff --git a/crates/frontend-metricsql/tests/lowering.rs b/crates/frontend-metricsql/tests/lowering.rs index 133fc328b..3add8ee2a 100644 --- a/crates/frontend-metricsql/tests/lowering.rs +++ b/crates/frontend-metricsql/tests/lowering.rs @@ -13,13 +13,13 @@ fn lower(query: &str) -> QueryExpr { #[test] fn selector_range_aggregate_and_call_share_the_canonical_shape() { let query = r#"sum by (job) (rate(http_requests_total{status=~"5.."}[5m]))"#; - let tree = lower(query); + let dag = lower(query); let QueryExpr::Aggregate { reduction, measures, child, .. - } = tree + } = dag else { panic!("expected outer aggregate"); }; @@ -43,13 +43,13 @@ fn selector_range_aggregate_and_call_share_the_canonical_shape() { #[test] fn default_rollup_with_explicit_range_is_last_over_time() { - let tree = lower("default_rollup(cpu_usage[5m])"); + let dag = lower("default_rollup(cpu_usage[5m])"); let QueryExpr::Aggregate { reduction, measures, child, .. - } = tree + } = dag else { panic!("expected aggregate"); }; diff --git a/crates/frontend-promql/src/error.rs b/crates/frontend-promql/src/error.rs index 1885a5f42..6f996b11d 100644 --- a/crates/frontend-promql/src/error.rs +++ b/crates/frontend-promql/src/error.rs @@ -1,12 +1,12 @@ use std::fmt; -use asap_types::pre_asap::ResolveTreeError; +use asap_types::pre_asap::ResolveDAGError; use asap_types::workload::WorkloadError; /// Errors from lowering a PromQL query (parse → the canonical, unresolved -/// tree, built directly → +/// DAG, built directly → /// [`resolve_root`](asap_types::pre_asap::resolve_root) binds it to the -/// resolved tree, issue #179). +/// resolved DAG, issue #179). /// /// Carries no DataFusion type — the PromQL front end never depends on the SQL /// stack. The language-neutral variants (`UnsupportedFeature` / `WrongLanguage` @@ -31,9 +31,9 @@ pub enum PromqlError { InvalidParameter(String), /// The workload's query language is not PromQL. WrongLanguage(String), - /// Resolving the canonical unresolved tree failed (name resolution + /// Resolving the canonical unresolved DAG failed (name resolution /// against the bound schema). - Convert(ResolveTreeError), + Convert(ResolveDAGError), } impl fmt::Display for PromqlError { @@ -54,8 +54,8 @@ impl fmt::Display for PromqlError { impl std::error::Error for PromqlError {} -impl From for PromqlError { - fn from(e: ResolveTreeError) -> Self { +impl From for PromqlError { + fn from(e: ResolveDAGError) -> Self { Self::Convert(e) } } diff --git a/crates/frontend-promql/src/lib.rs b/crates/frontend-promql/src/lib.rs index 60254d3d9..31b169359 100644 --- a/crates/frontend-promql/src/lib.rs +++ b/crates/frontend-promql/src/lib.rs @@ -4,7 +4,7 @@ //! Emits [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) itself — the //! canonical `QueryExpr`, generic over an unresolved //! [`ColumnRef`](asap_types::pre_asap::ColumnRef) — directly, rather than a -//! separate per-language relational tree; `resolve_root` runs the +//! separate per-language relational DAG; `resolve_root` runs the //! [`SchemaResolver`](asap_types::pre_asap::SchemaResolver) for positional name resolution. //! Depends on the PromQL parser only — never on the SQL / DataFusion stack. diff --git a/crates/frontend-promql/src/promql.rs b/crates/frontend-promql/src/promql.rs index e1061683f..83251053e 100644 --- a/crates/frontend-promql/src/promql.rs +++ b/crates/frontend-promql/src/promql.rs @@ -6,7 +6,7 @@ //! - **Lowering** builds *directly in canonical shape* here (issue #179): the //! walk interprets PromQL semantics (range vectors, aggregate operators, //! label matchers) and emits `UnresolvedQueryExpr` nodes with unresolved -//! `ColumnRef`s — the same tree shape +//! `ColumnRef`s — the same DAG shape //! [`resolve_root`](asap_types::pre_asap::resolve_root) later binds to //! canonical, positional `QueryExpr`. The structural decisions a //! separate converter stage would otherwise have to make (heavy-hitter @@ -17,7 +17,7 @@ //! the schema-*dependent* work: binding every `ColumnRef` to its //! positional `ColumnId`. //! -//! # PromQL → canonical unresolved-tree mapping (summary) +//! # PromQL → canonical unresolved-DAG mapping (summary) //! //! | PromQL | Canonical shape | //! |---|---| @@ -79,7 +79,7 @@ use crate::error::PromqlError as LoweringError; type Result = std::result::Result; -/// Parses and lowers (→ the canonical, unresolved tree) a PromQL query string. +/// Parses and lowers (→ the canonical, unresolved DAG) a PromQL query string. pub(crate) struct PromqlLowerer; #[derive(Debug, Clone)] @@ -357,8 +357,8 @@ fn walk(expr: &Expr) -> Result { /// `extract_matrix` can't accept it. Lower the sub-query recursively and reduce /// it per series (issue #27). fn walk_call(call: &Call) -> Result { - if let Some(tree) = range_fn_over_subquery(call)? { - return Ok(tree); + if let Some(dag) = range_fn_over_subquery(call)? { + return Ok(dag); } build(lower_inner_call(call)?, vec![], Outer::None) } @@ -491,7 +491,7 @@ fn walk_aggregate(agg: &AggregateExpr) -> Result { // exclusion form when the modifier was `without(...)`. let built = match lower_inner(&agg.expr) { Ok(inner) => build(inner, keys, outer)?, - Err(_) => build_over_subtree(outer, keys, walk(&agg.expr)?)?, + Err(_) => build_over_sub_dag(outer, keys, walk(&agg.expr)?)?, }; Ok(mark_without(built, without)) } @@ -555,13 +555,13 @@ fn outer_kind(agg: &AggregateExpr) -> Result { }) } -/// Wrap an already-lowered Unresolved subtree in the outer aggregation. This is the +/// Wrap an already-lowered Unresolved sub-DAG in the outer aggregation. This is the /// general-nesting counterpart to [`build`]: where `build` assembles the /// two-level shape from a flat [`Inner`], this composes the outer operator over /// an arbitrary child (`max(sum by (job) (…))`, `sum(a + b)`, …). /// /// A heavy-hitter `TopK` is only recognised on the flat `count_over_time` shape -/// (handled in `build`); over a general subtree, `topk`/`bottomk` is a generic +/// (handled in `build`); over a general sub-DAG, `topk`/`bottomk` is a generic /// order-by-value + limit — the same `Sort{partition_by} → Limit` pair `build` /// emits for any non-heavy-hitter ranking. /// Flip the outer `Aggregate` produced for a `without(...)` grouping into the @@ -576,7 +576,7 @@ fn outer_kind(agg: &AggregateExpr) -> Result { /// build this node) decides `PerEntity` vs `Reduce(by)` *without* knowing /// about `without` yet — it only ever sees `by`-mode keys, since `without`'s /// excluded-labels list is applied here, after the fact, exactly like the -/// pre-#179 legacy `relational::QueryExpr` tree's own `mark_without` did (its +/// pre-#179 legacy `relational::QueryExpr` DAG's own `mark_without` did (its /// converter read `without` only after this front-end step had already set /// it). Whether /// `reduction_for` picked `PerEntity` (only possible when `keys` was empty) @@ -584,11 +584,11 @@ fn outer_kind(agg: &AggregateExpr) -> Result { /// `Reduce(without(keys))`: a `without` grouping is never label-preserving — /// per-entity requires `!by.is_without()` — so this both re-tags an existing /// `Reduce` and upgrades a wrongly-early `PerEntity` guess, uniformly. -fn mark_without(tree: Unresolved, without: bool) -> Unresolved { +fn mark_without(dag: Unresolved, without: bool) -> Unresolved { if !without { - return tree; + return dag; } - match tree { + match dag { Unresolved::Aggregate { reduction, measures, @@ -614,7 +614,7 @@ fn mark_without(tree: Unresolved, without: bool) -> Unresolved { } } -fn build_over_subtree(outer: Outer, keys: Vec, child: Unresolved) -> Result { +fn build_over_sub_dag(outer: Outer, keys: Vec, child: Unresolved) -> Result { Ok(match outer { // `walk_aggregate` always passes a real aggregator; `None` can't occur. Outer::None => child, @@ -750,7 +750,7 @@ fn classic_histogram_quantile(q: f64, output_name: &str, child: Unresolved) -> U /// (`HistogramQuantile`), native histograms / raw samples take the sketch-able /// `Quantile` (issues #43 / #79) — so the two functions cannot diverge. /// -/// The vector argument is lowered once per branch, duplicating the subtree — +/// The vector argument is lowered once per branch, duplicating the sub-DAG — /// a future workload-level reuse pass could hoist it back into a single /// producer. /// @@ -906,7 +906,7 @@ fn is_presence_fn(name: &str) -> bool { /// `absent(v)` / `absent_over_time(m[w])` / `present_over_time(m[w])` — lowered /// to an `Aggregate{[Absent/…]}` over the (instant or range) argument. The /// empty-result → synthesized-1-sample logic is a post-ASAP/runtime concern; -/// the canonical tree only marks the operation (issue #47). +/// the canonical DAG only marks the operation (issue #47). fn walk_presence(call: &Call) -> Result { let func = match call.func.name { "absent" => AggIntent::Absent, @@ -1470,7 +1470,7 @@ fn lower_inner_call(call: &Call) -> Result { } } -/// Assemble the Layer-2 tree from a lowered inner vector, the resolved group +/// Assemble the Layer-2 DAG from a lowered inner vector, the resolved group /// keys, and the enclosing aggregator shape. fn build(inner: Inner, keys: Vec, outer: Outer) -> Result { match outer { @@ -1558,7 +1558,7 @@ fn build(inner: Inner, keys: Vec, outer: Outer) -> Result }; let additive_ranking = measure.is_supported(descending); if additive_ranking { - // Preserve the ranked aggregate intent in the canonical tree so the + // Preserve the ranked aggregate intent in the canonical DAG so the // intent algebra is explicit about what is being computed. // Post-ASAP binding may fuse the Count and TopK into a // single-pass heavy-hitter sketch (SpaceSaving / @@ -1622,7 +1622,7 @@ fn build(inner: Inner, keys: Vec, outer: Outer) -> Result /// Decide `PerEntity` vs `Reduce(by)` for a canonical `Aggregate`, entirely /// from local PromQL semantics: the keys and whether this operation preserves -/// each input series. It never infers entity reduction from the child tree's +/// each input series. It never infers entity reduction from the child DAG's /// temporal shape. `without()` is applied /// separately, post-hoc, by `mark_without` — see its doc for why that's still /// correct here. @@ -1676,7 +1676,7 @@ fn windowed_aggregate( } } -/// `Aggregate{reduction, [intent]}` directly over an existing Unresolved subtree — the +/// `Aggregate{reduction, [intent]}` directly over an existing Unresolved sub-DAG — the /// OUTER level of a two-level aggregation such as `sum(rate(…))` or the /// `Aggregate{[Quantile]}` that wraps a `histogram_quantile` argument. fn outer_aggregate( diff --git a/crates/frontend-promql/tests/histogram_metadata.rs b/crates/frontend-promql/tests/histogram_metadata.rs index 60ff5a60e..55f35ddeb 100644 --- a/crates/frontend-promql/tests/histogram_metadata.rs +++ b/crates/frontend-promql/tests/histogram_metadata.rs @@ -11,7 +11,7 @@ use asap_types::pre_asap::{AggIntent, QueryExpr}; use asap_types::types::AccuracyTarget; use support::{lower_promql, lower_promql_with_histograms}; -/// The histogram/quantile intent kind in the lowered tree: `"HQ"` for the +/// The histogram/quantile intent kind in the lowered DAG: `"HQ"` for the /// classic-bucket `HistogramQuantile`, `"Q"` for the sketch-able `Quantile`. fn quantile_kind(qe: &QueryExpr) -> &'static str { fn walk(e: &QueryExpr) -> Option<&'static str> { diff --git a/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs b/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs index 12afb94da..1b37cdee4 100644 --- a/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs +++ b/crates/frontend-promql/tests/observability/awesome_prometheus_alerts.rs @@ -5,7 +5,7 @@ //! host/hardware, node-exporter, databases, message brokers, Kubernetes, and //! more (`tests/data/awesome_prometheus_alerts.txt`). //! -//! We *lower* (parse → the canonical tree), we do not execute. Two guarantees: +//! We *lower* (parse → the canonical DAG), we do not execute. Two guarantees: //! 1. **Totality** — every real-world query returns `Ok` or a clean //! `LoweringError` and never panics. //! 2. **Parseability** — none of them fail at the *parse* stage; the private @@ -49,7 +49,7 @@ fn ok(q: &str) -> QueryExpr { .unwrap_or_else(|e| panic!("expected {q:?} to lower, got error: {e}")) } -/// Every `AggIntent` in the tree. +/// Every `AggIntent` in the DAG. fn intents(e: &QueryExpr) -> Vec { let mut out = Vec::new(); fn go(e: &QueryExpr, out: &mut Vec) { diff --git a/crates/frontend-promql/tests/observability/o11y_bench_promql.rs b/crates/frontend-promql/tests/observability/o11y_bench_promql.rs index e178d8981..cd371be69 100644 --- a/crates/frontend-promql/tests/observability/o11y_bench_promql.rs +++ b/crates/frontend-promql/tests/observability/o11y_bench_promql.rs @@ -12,7 +12,7 @@ //! verbatim into a differently-licensed test suite is fine as-is is an open //! question — tracked in issue #135, not resolved by this file's existence. //! -//! We *lower* (parse → the canonical tree), we do not execute. Totality: every query +//! We *lower* (parse → the canonical DAG), we do not execute. Totality: every query //! returns `Ok` or a clean `LoweringError` and never panics. Given the corpus //! is small and hand-picked from realistic incident-response queries, we also //! assert full lowering coverage — a regression here means a real pattern diff --git a/crates/frontend-promql/tests/observability/promql_corpus.rs b/crates/frontend-promql/tests/observability/promql_corpus.rs index f265edb45..1bc7e2166 100644 --- a/crates/frontend-promql/tests/observability/promql_corpus.rs +++ b/crates/frontend-promql/tests/observability/promql_corpus.rs @@ -27,7 +27,7 @@ use asap_types::pre_asap::query_expr::QueryExpr; use asap_types::types::AccuracyTarget; use support::lower_promql; -/// This crate has no "bind me one tree" public API any more — +/// This crate has no "bind me one DAG" public API any more — /// `SketchAlgorithmStrategy::replacements` always returns every candidate, and /// a caller decides what to keep. This test-only helper reproduces the /// take-the-first-(`cost_model`-preferred)-candidate pattern so [`bind_tally`] @@ -104,10 +104,10 @@ struct BindTally { fn bind_tally(corpus: &str, accuracy: AccuracyTarget) -> BindTally { let mut t = BindTally::default(); for q in queries(corpus) { - let Ok(tree) = lower_promql(q, accuracy.clone()) else { + let Ok(dag) = lower_promql(q, accuracy.clone()) else { continue; }; - match bind(&tree) { + match bind(&dag) { Ok(bound) if matches!(bound.expr, SummaryExpr::KeepPreAsap(_)) => t.unchanged += 1, Ok(_) => t.transformed += 1, Err(_) => t.errored += 1, diff --git a/crates/frontend-promql/tests/promql_binding_regressions.rs b/crates/frontend-promql/tests/promql_binding_regressions.rs index b5edf6049..26d1be06d 100644 --- a/crates/frontend-promql/tests/promql_binding_regressions.rs +++ b/crates/frontend-promql/tests/promql_binding_regressions.rs @@ -34,8 +34,8 @@ fn irate_and_rate_have_distinct_canonical_intents() { #[test] fn count_is_row_count_not_distinct_sample_value_count() { use asap_types::pre_asap::{AggIntent, QueryExpr}; - let tree = lower_promql("count(smoke_gauge)", AccuracyTarget::Exact).unwrap(); - let QueryExpr::Aggregate { measures, .. } = tree else { + let dag = lower_promql("count(smoke_gauge)", AccuracyTarget::Exact).unwrap(); + let QueryExpr::Aggregate { measures, .. } = dag else { panic!("expected aggregate") }; assert!(matches!(measures.as_slice(), [AggIntent::Count { .. }])); diff --git a/crates/frontend-promql/tests/promql_conformance.rs b/crates/frontend-promql/tests/promql_conformance.rs index 69fbb2230..a7b405f0d 100644 --- a/crates/frontend-promql/tests/promql_conformance.rs +++ b/crates/frontend-promql/tests/promql_conformance.rs @@ -1,8 +1,8 @@ -//! PromQL **semantic conformance** for the parse-to-canonical-tree lowering. +//! PromQL **semantic conformance** for the parse-to-canonical-DAG lowering. //! //! We *lower* PromQL to the intent algebra; we do not *execute* it. So "same //! semantic job as Prometheus" here means: for each canonical query, does the -//! canonical tree encode the **documented PromQL meaning** — and where we knowingly +//! canonical DAG encode the **documented PromQL meaning** — and where we knowingly //! diverge (reject, approximate, or drop a modifier), is that pinned by a test //! so it stays visible? //! @@ -55,11 +55,11 @@ fn ok(q: &str) -> QueryExpr { fn rejected(q: &str) -> LoweringError { match lower_promql(q, AccuracyTarget::Exact) { Err(e) => e, - Ok(tree) => panic!("expected {q:?} to be rejected, but it lowered to: {tree:?}"), + Ok(dag) => panic!("expected {q:?} to be rejected, but it lowered to: {dag:?}"), } } -/// Every `AggIntent` anywhere in the tree, root-to-leaf. +/// Every `AggIntent` anywhere in the DAG, root-to-leaf. fn intents(e: &QueryExpr) -> Vec { let mut out = Vec::new(); collect(e, &mut out); @@ -148,7 +148,7 @@ fn has bool>(e: &QueryExpr, pred: F) -> bool { intents(e).iter().any(pred) } -/// Whether the tree contains a `Mul`-by-`PromqlScalarBridge(-1)` anywhere — the shape unary +/// Whether the DAG contains a `Mul`-by-`PromqlScalarBridge(-1)` anywhere — the shape unary /// negation lowers to (issue #36). fn negates_via_scalar(e: &QueryExpr) -> bool { let is_neg_one = |q: &QueryExpr| { @@ -239,7 +239,7 @@ fn name_regex_matcher_is_rejected__GAP() { #[test] fn range_vector_selector_is_time_range() { // SEMANTICS: `[5m]` turns an instant vector into a range vector, - // represented in the canonical tree as a dedicated `TimeRange` node. + // represented in the canonical DAG as a dedicated `TimeRange` node. let qe = ok("node_cpu_seconds_total[5m]"); let QueryExpr::TimeRange { range, .. } = &qe else { panic!("expected TimeRange for a range-vector selector, got {qe:?}"); @@ -549,7 +549,7 @@ fn histogram_quantile_over_rate() { fn histogram_quantile_over_sum_by_le_preserves_le_grouping() { // SEMANTICS: the standard pattern — bucket rates summed by `le`, then the // quantile. The `sum by (le)` aggregation must survive into the - // canonical tree. + // canonical DAG. let qe = ok( "histogram_quantile(0.99, sum by(le) (rate(demo_api_request_duration_seconds_bucket[5m])))", ); @@ -635,7 +635,7 @@ fn unary_negation_lowers_as_multiply_by_minus_one() { "sum(-node_cpu_seconds_total)", ] { let qe = ok(q); - // A `Mul`-by-`-1` against a `PromqlScalarBridge(-1)` appears somewhere in every tree. + // A `Mul`-by-`-1` against a `PromqlScalarBridge(-1)` appears somewhere in every DAG. assert!( negates_via_scalar(&qe), "no `* -1` negation found in {q}: {qe:?}" @@ -859,7 +859,7 @@ fn outer_aggregate_over_nested_aggregate_nests() { // `max(sum by (job) (rate(m[5m])))` — an outer cross-series reduction over a // nested per-group reduction over a per-series rate: three stacked levels the // flat two-level template rejected. Each level survives into the - // canonical tree (issue #27). + // canonical DAG (issue #27). let qe = ok("max(sum by (job) (rate(http_requests_total[5m])))"); let QueryExpr::Aggregate { measures, child, .. @@ -955,7 +955,7 @@ fn outer_group_key_over_binary_op_resolves_on_both_sides() { // Issue #52: an outer aggregate's group key that appears in *neither* side of // a binary op — the metric-name label `__name__`, or a plain `job` — must // still resolve. Each `or` side is bound independently against its own - // sub-tree, so the key is seeded as an inherited column on both sides. + // sub-DAG, so the key is seeded as an inherited column on both sides. let qe = ok(r#"sum by (__name__)(metric_a{env="1"} or metric_b{env="2"})"#); let QueryExpr::Aggregate { reduction, child, .. @@ -1833,7 +1833,7 @@ fn no_arg_calendar_function_reads_the_eval_time() { #[test] fn timestamp_composes_under_an_outer_aggregation() { // `sum by (job) (timestamp(up))` — the per-series `timestamp` transform sits - // below an ordinary grouped sum. Both intents must appear in the tree. + // below an ordinary grouped sum. Both intents must appear in the DAG. let qe = ok("sum by (job) (timestamp(up))"); assert!(has(&qe, |i| *i == AggIntent::TimeFn(TimeFunc::Timestamp))); assert!(has(&qe, |i| matches!(i, AggIntent::Sum { .. }))); diff --git a/crates/frontend-promql/tests/promql_equivalence.rs b/crates/frontend-promql/tests/promql_equivalence.rs index d1177cc59..9d0cab2ad 100644 --- a/crates/frontend-promql/tests/promql_equivalence.rs +++ b/crates/frontend-promql/tests/promql_equivalence.rs @@ -1,10 +1,10 @@ -//! PromQL **semantic-equivalence proving** for the parse-to-canonical-tree lowering. +//! PromQL **semantic-equivalence proving** for the parse-to-canonical-DAG lowering. //! //! The lowering is a *normalizer*: it should map a whole class of -//! semantically-equivalent PromQL strings to **one** canonical tree, and +//! semantically-equivalent PromQL strings to **one** canonical DAG, and //! must keep semantically-*distinct* queries distinct. This suite proves: //! -//! 1. Equivalence classes collapse to an identical canonical tree (`assert_equiv`). +//! 1. Equivalence classes collapse to an identical canonical DAG (`assert_equiv`). //! 2. Distinct meanings stay distinct (`assert_distinct`). //! 3. The lowering never *wrongly* equates distinct semantics — the cases it //! cannot faithfully distinguish are **rejected**, not silently merged. @@ -26,25 +26,25 @@ fn lo(q: &str) -> QueryExpr { lower_promql(q, AccuracyTarget::Exact).unwrap_or_else(|e| panic!("{q:?} should lower: {e}")) } -/// Every member of an equivalence class must lower to the *same* canonical tree. +/// Every member of an equivalence class must lower to the *same* canonical DAG. fn assert_equiv(class: &[&str]) { let first = lo(class[0]); for q in &class[1..] { assert_eq!( lo(q), first, - "expected {q:?} ≡ {:?}, but they lowered to different trees", + "expected {q:?} ≡ {:?}, but they lowered to different DAGs", class[0] ); } } -/// Two semantically-distinct queries must lower to *different* canonical trees. +/// Two semantically-distinct queries must lower to *different* canonical DAGs. fn assert_distinct(a: &str, b: &str) { assert_ne!( lo(a), lo(b), - "{a:?} and {b:?} must not collapse to the same tree" + "{a:?} and {b:?} must not collapse to the same DAG" ); } @@ -79,7 +79,7 @@ fn whitespace_is_irrelevant() { #[test] fn label_matcher_order_is_equivalent() { - // A matcher set is unordered: same series, so same canonical tree (FIX: + // A matcher set is unordered: same series, so same canonical DAG (FIX: // predicates are now canonicalised by (name, value) at lowering time). assert_equiv(&[r#"up{job="a",env="prod"}"#, r#"up{env="prod",job="a"}"#]); } @@ -153,7 +153,7 @@ fn changes_and_resets_are_not_count_over_time() { // PromQL: count_over_time = #samples, changes = #value-changes, // resets = #counter-resets. They previously all collapsed to `Count`; now // each lowers to its own intent (issue #44), so all three are pairwise - // distinct canonical trees rather than being rejected or merged. + // distinct canonical DAGs rather than being rejected or merged. assert_distinct("changes(m[5m])", "count_over_time(m[5m])"); assert_distinct("resets(m[5m])", "count_over_time(m[5m])"); assert_distinct("changes(m[5m])", "resets(m[5m])"); @@ -163,7 +163,7 @@ fn changes_and_resets_are_not_count_over_time() { fn group_is_not_sum() { // PromQL `group` returns a constant 1 per group; it previously collapsed // onto `sum` (sum of values). It now lowers to its own `Group` intent - // (issue #49) — a distinct canonical tree from `sum`, not merged. + // (issue #49) — a distinct canonical DAG from `sum`, not merged. assert_distinct("group(up)", "sum(up)"); assert_distinct("group by (job) (up)", "sum by (job) (up)"); } diff --git a/crates/frontend-promql/tests/promql_lowering.rs b/crates/frontend-promql/tests/promql_lowering.rs index 9d3b80de9..619d066c2 100644 --- a/crates/frontend-promql/tests/promql_lowering.rs +++ b/crates/frontend-promql/tests/promql_lowering.rs @@ -1,4 +1,4 @@ -//! End-to-end tests for PromQL → unresolved → canonical tree lowering. +//! End-to-end tests for PromQL → unresolved → canonical DAG lowering. use std::time::Duration; @@ -56,14 +56,14 @@ fn distinct_over_time_preserves_cardinality_accuracy_and_nested_windows() { "sum by(job)(distinct_over_time(cpu_usage[5m]))", ] { for accuracy in [AccuracyTarget::Exact, AccuracyTarget::Epsilon(0.02)] { - let tree = lower_promql(query, accuracy.clone()).unwrap(); + let dag = lower_promql(query, accuracy.clone()).unwrap(); let mut intents = Vec::new(); - collect_intents(&tree, &mut intents); + collect_intents(&dag, &mut intents); assert!( intents.iter().any(|intent| matches!( intent, AggIntent::Cardinality { accuracy: actual, .. } if actual == &accuracy )), - "{query}: {tree:?}" + "{query}: {dag:?}" ); assert!(!intents .iter() @@ -264,7 +264,7 @@ fn histogram_quantile_over_sum_by_le_preserves_grouping() { // The canonical Prometheus histogram pattern. Previously returned // UnsupportedFeature because `extract_matrix` couldn't see through the // `sum by (le)` aggregate; now the `le` grouping survives into the - // canonical tree. + // canonical DAG. let qe = lower(r#"histogram_quantile(0.99, sum by (le) (rate(http_requests_bucket[5m])))"#); let QueryExpr::Aggregate { measures, child, .. @@ -486,19 +486,19 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { Reduction::by(vec![2]), ), ] { - let tree = lower(query); + let dag = lower(query); let QueryExpr::Aggregate { measures, reduction: actual, child, .. - } = &tree + } = &dag else { - panic!("expected outer Count: {tree:?}"); + panic!("expected outer Count: {dag:?}"); }; assert!( matches!(measures.as_slice(), [AggIntent::Count { .. }]), - "{query}: {tree:?}" + "{query}: {dag:?}" ); assert_eq!(actual, &reduction, "{query}"); let QueryExpr::Aggregate { @@ -508,11 +508,11 @@ fn count_over_distinct_over_time_preserves_both_aggregates() { .. } = child.as_ref() else { - panic!("expected inner per-series Cardinality: {tree:?}"); + panic!("expected inner per-series Cardinality: {dag:?}"); }; assert!( matches!(measures.as_slice(), [AggIntent::Cardinality { .. }]), - "{query}: {tree:?}" + "{query}: {dag:?}" ); assert_eq!(reduction, &Reduction::PerEntity, "{query}"); assert!( @@ -534,17 +534,17 @@ fn count_never_lowers_to_distinct_sample_values() { "count(count_over_time(up[5m]))", "count_over_time(up[5m])", ] { - let tree = lower(query); - let intents = all_intents(&tree); + let dag = lower(query); + let intents = all_intents(&dag); assert!( intents.iter().any(|i| matches!(i, AggIntent::Count { .. })), - "{query}: {tree:?}" + "{query}: {dag:?}" ); assert!( !intents .iter() .any(|i| matches!(i, AggIntent::Cardinality { .. })), - "{query}: {tree:?}" + "{query}: {dag:?}" ); } } @@ -829,7 +829,7 @@ fn binary_op_binds_each_branch_against_its_own_schema() { ); } -/// Collect every `AggIntent` in the tree, root-to-leaf. +/// Collect every `AggIntent` in the DAG, root-to-leaf. fn all_intents(e: &QueryExpr) -> Vec { let mut out = Vec::new(); collect_intents(e, &mut out); @@ -856,7 +856,7 @@ fn collect_intents(e: &QueryExpr, out: &mut Vec) { } } -/// True if any `AggIntent` anywhere in the tree satisfies `pred`. +/// True if any `AggIntent` anywhere in the DAG satisfies `pred`. fn has_intent bool>(e: &QueryExpr, pred: F) -> bool { all_intents(e).iter().any(pred) } @@ -1328,7 +1328,7 @@ fn histogram_quantiles_rejects_an_out_of_range_quantile() { } } -// A subquery's `offset`/`@` shift the whole subquery, so the tree keeps them. +// A subquery's `offset`/`@` shift the whole subquery, so the DAG keeps them. #[test] fn subquery_time_shift_is_retained() { let QueryExpr::Aggregate { child, .. } = lower("max_over_time(m[5m:1m] offset 1m)") else { diff --git a/crates/frontend-promql/tests/univmon_candidates.rs b/crates/frontend-promql/tests/univmon_candidates.rs index d34768e8f..1c7083464 100644 --- a/crates/frontend-promql/tests/univmon_candidates.rs +++ b/crates/frontend-promql/tests/univmon_candidates.rs @@ -8,7 +8,7 @@ use asap_aware_mapping::replacement::{default_strategies, search_workload_with_t use asap_aware_mapping::{Replacement, ReplacementStrategy, SketchAlgorithmStrategy, TargetSubDAG}; mod support; use asap_types::post_asap::{ - compile_post_asap_dag, cse::share_common_summary_subtrees, AccuracyError, BoundExpr, + compile_post_asap_dag, cse::share_common_summary_sub_dags, AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, ProbabilityExpr, ResultGuarantee, SketchAlgorithm, SketchQuery, SummaryExpr, SummaryFamilyType, SummaryInputExpr, SummaryNode, }; @@ -76,7 +76,7 @@ fn four_readouts_share_one_value_frequency_state_and_keep_honest_guarantees() { .enumerate() .map(|(id, (query, accuracy))| (id, candidate(query, accuracy))) .collect(); - let roots = share_common_summary_subtrees(roots); + let roots = share_common_summary_sub_dags(roots); let mut first_state = None; for (index, root) in &roots { let SummaryExpr::SummaryEstimate { diff --git a/crates/frontend-sql/src/error.rs b/crates/frontend-sql/src/error.rs index 5342f5a3f..04117352d 100644 --- a/crates/frontend-sql/src/error.rs +++ b/crates/frontend-sql/src/error.rs @@ -1,11 +1,11 @@ use std::fmt; -use asap_types::pre_asap::ResolveTreeError; +use asap_types::pre_asap::ResolveDAGError; /// Errors from lowering a SQL query (parse + plan via DataFusion → the -/// canonical, unresolved tree, built directly → +/// canonical, unresolved DAG, built directly → /// [`resolve_root`](asap_types::pre_asap::resolve_root) binds it to the -/// resolved tree, issue #179). +/// resolved DAG, issue #179). /// /// Carries no PromQL type — the SQL front end never depends on the PromQL /// parser. The language-neutral variants (`UnsupportedFeature` / `WrongLanguage` @@ -28,9 +28,9 @@ pub enum SqlError { UnsupportedFeature(String), /// The workload's query language is not SQL. WrongLanguage(String), - /// Resolving the canonical unresolved tree failed (name resolution + /// Resolving the canonical unresolved DAG failed (name resolution /// against the bound schema). - Convert(ResolveTreeError), + Convert(ResolveDAGError), } impl fmt::Display for SqlError { @@ -50,8 +50,8 @@ impl fmt::Display for SqlError { impl std::error::Error for SqlError {} -impl From for SqlError { - fn from(e: ResolveTreeError) -> Self { +impl From for SqlError { + fn from(e: ResolveDAGError) -> Self { Self::Convert(e) } } diff --git a/crates/frontend-sql/src/lib.rs b/crates/frontend-sql/src/lib.rs index 58474e2eb..4ec61e69d 100644 --- a/crates/frontend-sql/src/lib.rs +++ b/crates/frontend-sql/src/lib.rs @@ -4,7 +4,7 @@ //! Emits [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) itself — the //! canonical `QueryExpr`, generic over an unresolved //! [`ColumnRef`](asap_types::pre_asap::ColumnRef) — directly, rather than a -//! separate per-language relational tree; `resolve_root` runs the +//! separate per-language relational DAG; `resolve_root` runs the //! [`SchemaResolver`](asap_types::pre_asap::SchemaResolver) for positional name resolution. //! Depends on DataFusion only — never on the PromQL parser. @@ -24,7 +24,7 @@ pub use sql::{SqlCatalog, SqlLowerer}; /// /// The `catalog` supplies table schemas (used both to plan the SQL with /// DataFusion and to carry positional column identity into the resolved -/// tree). `accuracy` is threaded onto every approximate intent as it's built. +/// DAG). `accuracy` is threaded onto every approximate intent as it's built. pub async fn lower_sql( query: &str, catalog: &SqlCatalog, diff --git a/crates/frontend-sql/src/sql/collection_planning.rs b/crates/frontend-sql/src/sql/collection_planning.rs index 92fda701d..eba207751 100644 --- a/crates/frontend-sql/src/sql/collection_planning.rs +++ b/crates/frontend-sql/src/sql/collection_planning.rs @@ -164,7 +164,7 @@ impl ScalarUDFImpl for CollectionPlanningFunction { } } fn invoke_batch(&self, _args: &[ColumnarValue], _number_rows: usize) -> Result { - Err(DataFusionError::NotImplemented("collection planning adapter cannot execute; use a capable query engine or external exact subtree".into())) + Err(DataFusionError::NotImplemented("collection planning adapter cannot execute; use a capable query engine or external exact sub-DAG".into())) } } diff --git a/crates/frontend-sql/src/sql/expr.rs b/crates/frontend-sql/src/sql/expr.rs index 27ca85beb..06b682062 100644 --- a/crates/frontend-sql/src/sql/expr.rs +++ b/crates/frontend-sql/src/sql/expr.rs @@ -24,7 +24,7 @@ pub(super) fn split_conjuncts(expr: &Expr) -> Vec<&Expr> { } } -/// Translate a DataFusion `Expr` to the canonical, unresolved tree. +/// Translate a DataFusion `Expr` to the canonical, unresolved DAG. /// Returns `UnsupportedFeature` for anything not needed in v1. pub(super) fn df_expr_to_unresolved(expr: &Expr) -> Result { match expr { diff --git a/crates/frontend-sql/src/sql/mod.rs b/crates/frontend-sql/src/sql/mod.rs index e147e0bf3..52c065e9a 100644 --- a/crates/frontend-sql/src/sql/mod.rs +++ b/crates/frontend-sql/src/sql/mod.rs @@ -4,7 +4,7 @@ //! //! Parses SQL via DataFusion (over the catalog's registered tables), then //! walks the unoptimized `LogicalPlan` and emits `UnresolvedQueryExpr` nodes with -//! unresolved `ColumnRef`s directly (issue #179) — the same tree shape +//! unresolved `ColumnRef`s directly (issue #179) — the same DAG shape //! [`resolve_root`](asap_types::pre_asap::resolve_root) binds to canonical, //! positional `QueryExpr`. Unlike PromQL's front end, SQL's //! Ordinary SQL `Aggregate` nodes are `Reduction::Reduce`. The explicit @@ -114,7 +114,7 @@ fn current_accuracy() -> AccuracyTarget { /// Lowers SQL strings to the canonical [`UnresolvedQueryExpr`](asap_types::pre_asap::UnresolvedQueryExpr) /// over a table [`SqlCatalog`]. Call /// [`resolve_root`](asap_types::pre_asap::resolve_root) on the result for -/// the canonical, resolved tree. +/// the canonical, resolved DAG. pub struct SqlLowerer<'a> { catalog: &'a SqlCatalog, dialect: SqlDialect, @@ -2116,7 +2116,7 @@ fn expand_grouping_set(gs: &logical_expr::GroupingSet) -> Vec> { struct DerivedCols { cols: Vec>, /// Whether any column is genuinely derived. Without one the aggregate keeps - /// its original child, so trees that lower today keep their exact shape. + /// its original child, so DAGs that lower today keep their exact shape. any: bool, /// First same-name-different-value collision, reported only if the /// projection is actually inserted (see [`Self::wrap`]). @@ -2215,7 +2215,7 @@ impl DerivedCols { /// Wrap `input` in the materializing `Project`, or return it untouched when /// nothing needed deriving — so a query that lowers today keeps its exact - /// tree, and a name collision that the projection would have flattened only + /// DAG, and a name collision that the projection would have flattened only /// matters once the projection exists. fn wrap(self, input: Unresolved) -> Result { if !self.any { diff --git a/crates/frontend-sql/src/sql/types.rs b/crates/frontend-sql/src/sql/types.rs index 5382c4171..22611e6d0 100644 --- a/crates/frontend-sql/src/sql/types.rs +++ b/crates/frontend-sql/src/sql/types.rs @@ -1,6 +1,6 @@ //! Type bridges between DataFusion's Arrow types and the canonical `DataType`, plus //! the SQL table catalog used to register tables with DataFusion and to carry -//! resolved leaf schemas into the canonical, unresolved tree. +//! resolved leaf schemas into the canonical, unresolved DAG. use std::collections::HashMap; diff --git a/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs b/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs index 273597e7d..189ff0ff7 100644 --- a/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs +++ b/crates/frontend-sql/tests/data_quality_check/synthetic_packet_trace.rs @@ -65,9 +65,9 @@ fn queries() -> Vec { .collect() } -// ── tree helpers ────────────────────────────────────────────────────────────── +// ── DAG helpers ────────────────────────────────────────────────────────────── -/// Every `AggIntent` in the tree, root-to-leaf. +/// Every `AggIntent` in the DAG, root-to-leaf. fn intents(e: &QueryExpr) -> Vec { let mut out = Vec::new(); fn go(e: &QueryExpr, out: &mut Vec) { @@ -114,7 +114,7 @@ fn intents(e: &QueryExpr) -> Vec { | QueryExpr::CurrentTimestamp => {} // Scalar expression variants (issue #205): `AggIntent` only ever // lives in `Aggregate.measures`, never nested inside a scalar - // expression tree, so there's nothing to recurse into here. + // expression DAG, so there's nothing to recurse into here. QueryExpr::Column(_) | QueryExpr::Literal(_) | QueryExpr::Compare { .. } diff --git a/crates/frontend-sql/tests/maintained_population.rs b/crates/frontend-sql/tests/maintained_population.rs index 0019b3b83..5d73d7e90 100644 --- a/crates/frontend-sql/tests/maintained_population.rs +++ b/crates/frontend-sql/tests/maintained_population.rs @@ -5,7 +5,7 @@ use asap_types::{ post_asap::{ compile_post_asap_dag, maintained_population::{MaintainedPopulation, PopulationInput}, - share_common_summary_subtrees, SummaryExpr, ValueOperation, + share_common_summary_sub_dags, SummaryExpr, ValueOperation, }, pre_asap::{Column, DataType, QueryExpr, Schema}, types::AccuracyTarget, @@ -59,7 +59,7 @@ async fn sql_quantiles_share_rows_without_promql_lookback() { aggregate("SELECT approx_percentile_cont(latency, 0.99) FROM samples").await, ]; let rule = MaintainedPopulationStrategy::new(&roots); - let plans = share_common_summary_subtrees( + let plans = share_common_summary_sub_dags( roots .iter() .enumerate() @@ -120,7 +120,7 @@ async fn sql_scalar_readouts_share_membership() { roots.push(aggregate(&format!("SELECT {function} FROM samples")).await); } let rule = MaintainedPopulationStrategy::new(&roots); - let plans = share_common_summary_subtrees( + let plans = share_common_summary_sub_dags( roots .iter() .enumerate() @@ -167,7 +167,7 @@ async fn sql_topk_limits_share_maximum_k() { aggregate("SELECT * FROM samples ORDER BY latency DESC LIMIT 5").await, ]; let rule = MaintainedPopulationStrategy::new(&roots); - let plans = share_common_summary_subtrees( + let plans = share_common_summary_sub_dags( roots .iter() .enumerate() diff --git a/crates/frontend-sql/tests/netflow/netflow.rs b/crates/frontend-sql/tests/netflow/netflow.rs index 09d742b8b..dbe315381 100644 --- a/crates/frontend-sql/tests/netflow/netflow.rs +++ b/crates/frontend-sql/tests/netflow/netflow.rs @@ -303,7 +303,7 @@ fn visit(qe: &QueryExpr, f: &mut impl FnMut(&QueryExpr)) { | QueryExpr::EvalTimestamp | QueryExpr::CurrentTimestamp => {} // Scalar expression variants (issue #205) aren't relational nodes; - // this visitor only walks the relational tree, so stop here. + // this visitor only walks the relational DAG, so stop here. QueryExpr::Column(_) | QueryExpr::Literal(_) | QueryExpr::Compare { .. } diff --git a/crates/frontend-sql/tests/sql_lowering.rs b/crates/frontend-sql/tests/sql_lowering.rs index 80f25fbc1..e3e065e23 100644 --- a/crates/frontend-sql/tests/sql_lowering.rs +++ b/crates/frontend-sql/tests/sql_lowering.rs @@ -1,9 +1,9 @@ -//! End-to-end SQL → unresolved → canonical tree lowering tests (positional IR). +//! End-to-end SQL → unresolved → canonical DAG lowering tests (positional IR). //! //! Validates the DataFusion front end: SQL parses + plans, lowers directly to //! the canonical, unresolved shape (`QueryExpr`, issue #179), and //! the shared `resolve_root` produces the positional, resolved canonical -//! tree (the same resolver the PromQL path uses). +//! DAG (the same resolver the PromQL path uses). use asap_frontend_sql::{lower_sql, lower_sql_dialect, SqlCatalog, SqlError as LoweringError}; use asap_types::pre_asap::schema::{Column, DataType, Schema}; @@ -235,7 +235,7 @@ async fn select_star_with_where_folds_predicate_onto_scan() { async fn multi_aggregate_group_by_binds_columns_positionally() { // SUM(bytes)=col 3, AVG(latency)=col 2, GROUP BY service=col 1. let qe = lower("SELECT service, SUM(bytes), AVG(latency) FROM metrics GROUP BY service").await; - let (by, measures) = find_aggregate(&qe).expect("expected an Aggregate in the tree"); + let (by, measures) = find_aggregate(&qe).expect("expected an Aggregate in the DAG"); assert_eq!(by, &vec![1], "GROUP BY service → column 1"); assert!( measures.contains(&AggIntent::Sum { col: Some(3) }), @@ -330,7 +330,7 @@ async fn count_ranked_topk_is_heavy_hitter() { async fn count_ranked_topk_via_alias_is_also_heavy_hitter() { // Regression for #20: aliasing `COUNT(*)` in the ORDER BY used to defeat the // SQL front-end gate. The positional `canonicalize` pass now promotes it too, - // so the aliased and inline forms produce an identical canonical tree. + // so the aliased and inline forms produce an identical canonical DAG. let inline = lower( "SELECT service, COUNT(*) FROM metrics GROUP BY service ORDER BY COUNT(*) DESC LIMIT 10", ) @@ -441,7 +441,7 @@ async fn inner_join_lowers_to_join_over_two_scans() { FROM metrics JOIN hosts ON metrics.service = hosts.service", ) .await; - let join = find_join(&qe).expect("expected a Join in the tree"); + let join = find_join(&qe).expect("expected a Join in the DAG"); let QueryExpr::Join { kind, left, right, .. } = join @@ -489,7 +489,7 @@ async fn join_predicate_disambiguates_shared_column_name() { FROM metrics JOIN hosts ON metrics.service = hosts.service", ) .await; - let join = find_join(&qe).expect("expected a Join in the tree"); + let join = find_join(&qe).expect("expected a Join in the DAG"); assert_eq!( join_eq_columns(join), [1, 4], @@ -510,7 +510,7 @@ async fn derived_table_join_disambiguates_via_alias() { JOIN (SELECT service, region FROM hosts) b ON a.service = b.service", ) .await; - let join = find_join(&qe).expect("expected a Join in the tree"); + let join = find_join(&qe).expect("expected a Join in the DAG"); assert_eq!( join_eq_columns(join), [0, 2], @@ -528,7 +528,7 @@ async fn derived_table_select_star_join_disambiguates_via_alias() { ON a.service = b.service", ) .await; - let join = find_join(&qe).expect("expected a Join in the tree"); + let join = find_join(&qe).expect("expected a Join in the DAG"); let [l, r] = join_eq_columns(join); assert_ne!( l, r, @@ -546,7 +546,7 @@ async fn self_join_disambiguates_via_aliases() { FROM metrics a JOIN metrics b ON a.service = b.service", ) .await; - let join = find_join(&qe).expect("expected a self-Join in the tree"); + let join = find_join(&qe).expect("expected a self-Join in the DAG"); assert_eq!( join_eq_columns(join), [1, 5], @@ -938,7 +938,7 @@ async fn groups_frame_is_rejected() { // ── Nested query functions: derived tables / inline views (issue #27) ─────────── -/// Collect every `AggIntent` in the tree, root-to-leaf. +/// Collect every `AggIntent` in the DAG, root-to-leaf. fn all_intents(qe: &QueryExpr) -> Vec { let mut out = Vec::new(); fn go(qe: &QueryExpr, out: &mut Vec) { @@ -981,7 +981,7 @@ fn all_intents(qe: &QueryExpr) -> Vec { async fn derived_table_aggregate_over_aggregate_nests() { // `MAX(s)` over a derived table `(SELECT service, SUM(bytes) AS s … GROUP BY // service)` — the SQL counterpart of PromQL function nesting (issue #27). - // Both reductions survive into the canonical tree: an outer `Max` over + // Both reductions survive into the canonical DAG: an outer `Max` over // the inner `Sum`. let qe = lower( "SELECT MAX(s) FROM \ @@ -997,7 +997,7 @@ async fn derived_table_aggregate_over_aggregate_nests() { intents.iter().any(|i| matches!(i, AggIntent::Sum { .. })), "inner SUM survives, got {intents:?}" ); - // The whole nested tree's output schema derives without error (positional + // The whole nested DAG's output schema derives without error (positional // resolution is total across the derived-table boundary). assert_eq!(qe.output_schema().unwrap().columns.len(), 1); } @@ -1185,8 +1185,8 @@ async fn count_distinct_carries_its_input_column() { #[tokio::test] async fn quantile_and_count_distinct_over_an_expression_bind_the_derived_column() { // A SQL aggregate has no "sample value" to fall back on, so an expression - // argument must never reach the canonical tree as `col: None` (#115). - // Since #110 it reaches the canonical tree as `col: Some(derived)` + // argument must never reach the canonical DAG as `col: None` (#115). + // Since #110 it reaches the canonical DAG as `col: Some(derived)` // instead of being rejected. for q in [ "SELECT approx_percentile_cont(bytes * 8, 0.95) FROM metrics", @@ -1333,7 +1333,7 @@ async fn time_bucketing_keeps_the_scan_predicate() { #[tokio::test] async fn a_plain_group_by_inserts_no_projection() { - // Queries that lowered before #110 must keep their exact tree shape — the + // Queries that lowered before #110 must keep their exact DAG shape — the // projection appears only when something actually needs materializing. for q in [ "SELECT service, SUM(bytes) FROM metrics GROUP BY service", diff --git a/crates/integration-tests/src/lib.rs b/crates/integration-tests/src/lib.rs index 99fb24226..59602ac07 100644 --- a/crates/integration-tests/src/lib.rs +++ b/crates/integration-tests/src/lib.rs @@ -8,7 +8,7 @@ //! are in scope. //! //! `fixtures` provides column/schema constructors used across test files. -//! Expected IR trees are always hand-constructed inside each test — nothing +//! Expected IR DAGs are always hand-constructed inside each test — nothing //! here derives or computes expected outputs. pub mod fixtures { diff --git a/crates/integration-tests/tests/cse.rs b/crates/integration-tests/tests/cse.rs index 094ffd658..bb11eee2d 100644 --- a/crates/integration-tests/tests/cse.rs +++ b/crates/integration-tests/tests/cse.rs @@ -2,13 +2,13 @@ //! #223). //! //! Drives the full staged pipeline this issue lands: two independently -//! lowered `QueryExpr` trees → `share_common_subtrees` (stage 1, +//! lowered `QueryExpr` DAGs → `share_common_sub_dags` (stage 1, //! `asap-types::pre_asap::cse`, run internally by `search_workload`) → //! `search_workload` (stage 2, `asap-aware-mapping`) — and asserts the //! sharing that stage 1 decides survives into stage 2's discovered //! `CandidateLogicalASAPDAGs` as one genuinely shared `TargetSubDAGCandidates`, not just one shared //! `Rc`. This is the "real caller" the issue's landing plan -//! requires before `share_common_subtrees` is allowed to exist at all (its +//! requires before `share_common_sub_dags` is allowed to exist at all (its //! predecessor, `asap-plan::cse::dedupe_subtrees`, was deleted in #192 for //! being unwired dead code). //! @@ -32,7 +32,7 @@ use asap_types::types::AccuracyTarget; /// Two workload entries that happen to submit the exact same query (a /// realistic case — two dashboards, or a query fired both standalone and as /// part of a larger batch) collapse onto one shared `Rc` after -/// `search_workload`'s internal `share_common_subtrees` pass, and onto one +/// `search_workload`'s internal `share_common_sub_dags` pass, and onto one /// genuinely-shared [`TargetSubDAGCandidates`](asap_aware_mapping::TargetSubDAGCandidates) — carrying /// every candidate discovered for it exactly once, not once per root — no /// second structural-equality pass at the post-ASAP layer needed for this @@ -40,7 +40,7 @@ use asap_types::types::AccuracyTarget; #[test] fn duplicate_workload_queries_collapse_onto_one_memo_group() { // Grouped (`by (job)`), so the shared `Aggregate`'s output schema carries - // a provable unique key — the legality gate `share_common_subtrees` + // a provable unique key — the legality gate `share_common_sub_dags` // enforces (see `asap-types::pre_asap::cse`'s module doc) — and its // `ExactAggregate(Sum)` realization is deterministic regardless of the // accuracy target, so this pins the sharing mechanism itself rather than @@ -51,7 +51,7 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { // Independently lowered: not yet sharing any `Rc`, even though they are // structurally identical (`resolve_root` gives each call its own fresh - // tree). + // DAG). assert_eq!( a, b, "fixture sanity: identical query text lowers identically" @@ -60,7 +60,7 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { let space = search_workload(vec![("a", Rc::new(a)), ("b", Rc::new(b))]); // roots[0] and roots[1] must have merged onto the same Rc — the - // `share_common_subtrees` pass `search_workload` runs internally. + // `share_common_sub_dags` pass `search_workload` runs internally. assert!( Rc::ptr_eq(&space.roots[0].1, &space.roots[1].1), "search_workload must collapse the two identical roots onto one Rc" @@ -68,7 +68,7 @@ fn duplicate_workload_queries_collapse_onto_one_memo_group() { // The single shared root is one discovered TargetSubDAG, holding one // TargetSubDAGCandidates with consumer_count 2 — SketchAlgorithmStrategy's one - // ExactAggregate candidate *and* SharedSubtreeStrategy's share-vs- + // ExactAggregate candidate *and* SharedSubDAGStrategy's share-vs- // recompute pair, exactly as `shared_aggregate_across_two_roots_gets_both_strategies_candidates` // (asap-aware-mapping::replacement's own equivalent, internal test) // pins for the same fixture shape. @@ -141,7 +141,7 @@ fn distinct_workload_queries_get_independent_memo_groups() { /// Single-query CSE (a repeated sub-expression within one query) also /// survives through `search_workload`: the two grouped-`Aggregate` branches /// of a `BinaryOp` collapse to one shared `Rc` in the internal -/// `share_common_subtrees` pass, and to one shared `TargetSubDAGCandidates` (with +/// `share_common_sub_dags` pass, and to one shared `TargetSubDAGCandidates` (with /// `consumer_count == 2`, one per branch) here. #[test] fn single_query_repeated_subexpression_shares_one_memo_group() { diff --git a/crates/integration-tests/tests/exact_composition.rs b/crates/integration-tests/tests/exact_composition.rs index fe72aea26..7e2e7c12a 100644 --- a/crates/integration-tests/tests/exact_composition.rs +++ b/crates/integration-tests/tests/exact_composition.rs @@ -768,8 +768,8 @@ fn dag_export_carries_explicit_stage_and_plain_schema_for_a_composed_plan() { .assemble_selected_dag(root) .unwrap() .unwrap(); - let graph = dag_export::export_summary(&composed); - let node = &graph.nodes[graph.root as usize]; + let dag = dag_export::export_summary(&composed); + let node = &dag.nodes[dag.root as usize]; assert_eq!(node.kind, "ValueOperation"); assert_eq!(node.detail["timing"], "query_time"); assert!(node.detail["operation"] diff --git a/crates/integration-tests/tests/kll_pane_execution.rs b/crates/integration-tests/tests/kll_pane_execution.rs index e818796fd..52eea7e63 100644 --- a/crates/integration-tests/tests/kll_pane_execution.rs +++ b/crates/integration-tests/tests/kll_pane_execution.rs @@ -2,8 +2,8 @@ mod physical_common; use asap_physical_operators::{ operators::{Operator, ReadoutQuery}, - physical_planner::{CompiledPhysicalDag, InputContract, Source}, - plan::{PhysicalDag, PhysicalOperator, PlanProperties}, + physical_planner::{CompiledPhysicalDAG, InputContract, Source}, + plan::{PhysicalDAG, PhysicalOperator, PlanProperties}, runtime::{Input, Limits, OutputStream, RunContext, Scope}, summary_kernels::datasketches_kll::DatasketchesKLLAccumulator, values::{Batch, Schema, Value}, @@ -47,11 +47,11 @@ fn query_scope() -> Scope { revision: 1, } } -fn fixture() -> (CompiledPhysicalDag, Operator, Schema) { +fn fixture() -> (CompiledPhysicalDAG, Operator, Schema) { let raw = raw_schema(); let build = Operator::summary_build(raw.clone(), family(200), 0, None, vec![]).unwrap(); let state = build.schema(); - let maintenance = CompiledPhysicalDag::from_operators( + let maintenance = CompiledPhysicalDAG::from_operators( BTreeMap::from([(0, InputContract::bounded(raw))]), BTreeMap::from([(1, (vec![0], build))]), vec![1], @@ -60,7 +60,7 @@ fn fixture() -> (CompiledPhysicalDag, Operator, Schema) { let merge = Operator::summary_merge(state.clone(), 0, vec![]).unwrap(); (maintenance, merge, state) } -fn pane_state(maintenance: &CompiledPhysicalDag, pane: i64) -> Arc { +fn pane_state(maintenance: &CompiledPhysicalDAG, pane: i64) -> Arc { // Twenty samples in each (start,end] one-minute pane; k=200 avoids // compaction so quantiles and sample counts have deterministic oracles. let raw = raw_schema(); @@ -139,7 +139,7 @@ fn five_panes_roundtrip_and_shared_merge_runs_once() { let (maintenance, merge, schema) = fixture(); let panes: Vec<_> = (0..6).map(|pane| pane_state(&maintenance, pane)).collect(); drop(maintenance); - let compiled = CompiledPhysicalDag::from_operators( + let compiled = CompiledPhysicalDAG::from_operators( (0..5) .map(|id| (id, InputContract::bounded(schema.clone()))) .collect(), @@ -224,7 +224,7 @@ fn five_panes_roundtrip_and_shared_merge_runs_once() { assert!((value(1) - (50 + offset * 20) as f64).abs() <= 1.); assert!((value(2) - (99 + offset * 20) as f64).abs() <= 1.); let starts = Arc::new(AtomicUsize::new(0)); - let mut dag = PhysicalDag::default(); + let mut dag = PhysicalDAG::default(); dag.add( 0, vec![], @@ -267,7 +267,7 @@ fn panes_reject_parameters_schema_and_missing_binding() { state: Arc::new(DatasketchesKLLAccumulator::new(128)), }; assert!(Batch::try_new(schema.clone(), vec![vec![wrong]]).is_err()); - let compiled = CompiledPhysicalDag::from_operators( + let compiled = CompiledPhysicalDAG::from_operators( BTreeMap::from([(0, InputContract::bounded(schema))]), BTreeMap::from([(1, (vec![0], merge))]), vec![1], diff --git a/crates/integration-tests/tests/nested.rs b/crates/integration-tests/tests/nested.rs index af25af3be..1e3e34ebd 100644 --- a/crates/integration-tests/tests/nested.rs +++ b/crates/integration-tests/tests/nested.rs @@ -101,13 +101,13 @@ fn q23_sum_by_job_over_filtered_scan() { ); } -// #25 — binary op over two complex subtrees +// #25 — binary op over two complex sub-DAGs // LHS: sum by (job) over rate over filtered scan // schema [ts, value, job, status]; outer by=[2] (job) // RHS: sum by (job) over rate over bare scan // schema [ts, value, job]; outer by=[2] (job) #[test] -fn q25_div_over_complex_subtrees() { +fn q25_div_over_complex_sub_dags() { let lhs_scan = QueryExpr::Scan { source: Source::TimeSeries { metric: "http_requests_total".into(), @@ -229,7 +229,7 @@ fn q53_outer_group_key_absent_from_nested_aggregate() { // #52 — an outer group key referenced by neither binary-op side (`__name__`) // still resolves. Each `or` side is bound independently against its own -// sub-tree, so `__name__` is seeded as an inherited column on both. Each side +// sub-DAG, so `__name__` is seeded as an inherited column on both. Each side // references only `env` (its matcher), so its schema is [ts, value, env, // __name__] (referenced `env` first, inherited `__name__` appended) → the // outer `by (__name__)` resolves to col 3 on both sides. The `or` carries the diff --git a/crates/integration-tests/tests/physical_common/mod.rs b/crates/integration-tests/tests/physical_common/mod.rs index aca4d4efa..93bebd338 100644 --- a/crates/integration-tests/tests/physical_common/mod.rs +++ b/crates/integration-tests/tests/physical_common/mod.rs @@ -1,6 +1,6 @@ use asap_physical_operators::{ operators::Operator, - physical_planner::{CompiledPhysicalDag, Source}, + physical_planner::{CompiledPhysicalDAG, Source}, runtime::{Limits, RunContext, Scope}, values::Batch, }; @@ -8,7 +8,7 @@ use futures::{executor::block_on, StreamExt}; use std::collections::BTreeMap; pub fn execute( - plan: &CompiledPhysicalDag, + plan: &CompiledPhysicalDAG, inputs: BTreeMap, scope: Scope, ) -> Vec> { diff --git a/crates/integration-tests/tests/precompute_raw_samples.rs b/crates/integration-tests/tests/precompute_raw_samples.rs index 2df295725..c348eed9f 100644 --- a/crates/integration-tests/tests/precompute_raw_samples.rs +++ b/crates/integration-tests/tests/precompute_raw_samples.rs @@ -1,4 +1,4 @@ -//! Planner-selected summaries over raw samples compile as precompute graphs +//! Planner-selected summaries over raw samples compile as precompute DAGs //! and produce the same estimates as feeding their kernel sample by sample. use std::{collections::BTreeMap, collections::BTreeSet, rc::Rc, sync::Arc}; @@ -18,7 +18,7 @@ use asap_physical_operators::{ AggregateCore, KeyByLabelValues, Statistic, }; use asap_types::post_asap::{ - compile_post_asap_dag, EntityIdentity, ExactKind, PostAsapDag, PostAsapOperatorPayload, + compile_post_asap_dag, EntityIdentity, ExactKind, PostAsapDAG, PostAsapOperatorPayload, SketchAlgorithm, SketchQuery, SummaryFamilyType, SummaryInputExpr, SummaryNode, SummaryUpdate, }; use asap_types::pre_asap::{expr_ir::ColumnRef, query_expr::Reduction}; @@ -74,7 +74,7 @@ fn candidates(query: &str, accuracy: AccuracyTarget) -> Vec> { } /// Raw-input summary nodes: `(dag, raw source id, summary id)`. -fn raw_summaries(dag: &PostAsapDag) -> Vec<(u64, u64)> { +fn raw_summaries(dag: &PostAsapDAG) -> Vec<(u64, u64)> { dag.nodes .iter() .filter(|node| matches!(node.payload, PostAsapOperatorPayload::SummaryAgg { .. })) @@ -113,7 +113,7 @@ fn samples() -> Vec<(Series, i64, f64)> { } fn execute( - dag: &PostAsapDag, + dag: &PostAsapDAG, source: u64, root: u64, rows: &[(Series, i64, f64)], @@ -128,7 +128,7 @@ fn execute( ); }); let program = serde_json::from_slice::< - asap_physical_operators::physical_planner::CompiledPhysicalDag, + asap_physical_operators::physical_planner::CompiledPhysicalDAG, >(&serde_json::to_vec(&program).unwrap()) .unwrap(); let schema = precompute::raw_sample_schema(); @@ -143,7 +143,7 @@ fn execute( source, Box::new(Operator::source(schema, vec![batch]).unwrap()) as Source<'_>, )]); - let graph = program.instantiate(sources).unwrap(); + let physical_dag = program.instantiate(sources).unwrap(); let context = RunContext::new( Scope::Ingestion { window_start_ms: 0, @@ -154,7 +154,10 @@ fn execute( ) .unwrap(); block_on(async { - let mut stream = graph.execute(program.roots(), context).unwrap().remove(0); + let mut stream = physical_dag + .execute(program.roots(), context) + .unwrap() + .remove(0); let mut result = Vec::new(); while let Some(batch) = stream.next().await { for row in batch.unwrap().rows() { @@ -273,7 +276,7 @@ fn readouts(state: &dyn AggregateCore, family: &SummaryFamilyType) -> Vec { /// or the family when it has no native state. fn check( query: &str, - dag: &PostAsapDag, + dag: &PostAsapDAG, source: u64, root: u64, rows: &[(Series, i64, f64)], @@ -455,7 +458,7 @@ fn raw_sample_summaries_compile_and_match_their_kernels() { /// Replace the raw summary of `sum by (service) (sum_over_time(m[5m]))` with /// another update, keeping its raw input and reduction. -fn grouped_raw_summary(family: SummaryFamilyType, input: SummaryUpdate) -> (PostAsapDag, u64, u64) { +fn grouped_raw_summary(family: SummaryFamilyType, input: SummaryUpdate) -> (PostAsapDAG, u64, u64) { let candidate = candidates( "sum by (service) (sum_over_time(m[5m]))", AccuracyTarget::Exact, diff --git a/crates/integration-tests/tests/promql_to_post_asap.rs b/crates/integration-tests/tests/promql_to_post_asap.rs index 1a8d364e0..73f781443 100644 --- a/crates/integration-tests/tests/promql_to_post_asap.rs +++ b/crates/integration-tests/tests/promql_to_post_asap.rs @@ -29,7 +29,7 @@ use asap_types::pre_asap::query_expr::{QueryExpr, Reduction}; use asap_types::pre_asap::schema::DataType; use asap_types::types::AccuracyTarget; -/// This crate has no "bind me one tree" public API any more — +/// This crate has no "bind me one DAG" public API any more — /// `SketchAlgorithmStrategy::replacements` always returns every candidate, and /// a caller decides what to keep. This test-only helper reproduces the /// take-the-first-(`cost_model`-preferred)-candidate pattern so the @@ -643,7 +643,7 @@ fn ddsketch_quantile_ratio_meets_the_shared_relative_error_target() { ); let shared = - asap_types::post_asap::share_common_summary_subtrees(vec![("ratio", node.clone())]); + asap_types::post_asap::share_common_summary_sub_dags(vec![("ratio", node.clone())]); let SummaryExpr::BinaryOp { lhs, rhs, .. } = &shared[0].1.expr else { panic!("expected binary ratio") }; @@ -892,7 +892,7 @@ fn planner_heap_topk_reference_execution_matches_ground_truth() { /// └─ KeepPreAsap(TimeRange{5m} → Scan) → {ts, value} /// ``` /// -/// The nested tree exercises both realizations: the approximate quantile +/// The nested DAG exercises both realizations: the approximate quantile /// binds a KLL sketch + readout; the per-series `rate` binds the exact /// counter-reset-aware accumulator (no estimate — its state is the value). #[test] @@ -1011,7 +1011,7 @@ fn promql_quantile_of_rate_binds_kll_over_rate_accumulator() { /// An exact workload binds zero sketches: `sum by (job) (m)` at /// `AccuracyTarget::Exact` still gets its mergeable exact accumulator, and -/// `avg(m)` (non-mergeable) passes through as a whole logical subtree. +/// `avg(m)` (non-mergeable) passes through as a whole logical sub-DAG. #[test] fn promql_exact_workload_binds_accumulators_not_sketches() { let pre_asap = lower_promql("sum by (job) (http_requests_total)", AccuracyTarget::Exact) diff --git a/crates/integration-tests/tests/schema.rs b/crates/integration-tests/tests/schema.rs index 6024619cc..7d7e2570a 100644 --- a/crates/integration-tests/tests/schema.rs +++ b/crates/integration-tests/tests/schema.rs @@ -1,7 +1,7 @@ //! `Schema::closed` propagation — open/closed invariant tests. //! //! Verifies that `QueryExpr::output_schema()` propagates the open/closed -//! completeness flag correctly through a lowered query tree. +//! completeness flag correctly through a lowered query DAG. //! //! Key invariant: a PromQL scan is always `closed: false` (open) because its //! label set is runtime-only. The schema freezes to `closed: true` exactly at diff --git a/crates/integration-tests/tests/sql_to_post_asap.rs b/crates/integration-tests/tests/sql_to_post_asap.rs index d2d418222..5d10fb2f5 100644 --- a/crates/integration-tests/tests/sql_to_post_asap.rs +++ b/crates/integration-tests/tests/sql_to_post_asap.rs @@ -11,9 +11,9 @@ //! //! `lower_promql` returns a *bare* `QueryExpr::Aggregate` for a top-level //! aggregation (`sum by (job) (m)`, `quantile(0.99, …)`), so [`realize`] can -//! bind it directly at the tree root. `lower_sql` never does: DataFusion's +//! bind it directly at the DAG root. `lower_sql` never does: DataFusion's //! planner always wraps even a single, unaliased aggregate in an identity -//! `Project` (confirmed below), so a SQL tree's *root* is normally `Project { +//! `Project` (confirmed below), so a SQL DAG's *root* is normally `Project { //! child: Aggregate { .. } }`. Final materialization retains that projection //! as a query-time value operation and independently plans its child, keeping //! both SELECT-list semantics and the summary-bound aggregate visible. @@ -37,7 +37,7 @@ use asap_types::pre_asap::schema::{Column, DataType, Schema}; use asap_types::types::AccuracyTarget; use asap_types::workload::SqlDialect; -/// This crate has no "bind me one tree" public API any more — +/// This crate has no "bind me one DAG" public API any more — /// `SketchAlgorithmStrategy::replacements` always returns every candidate, and /// a caller decides what to keep. This test-only helper reproduces the /// take-the-first-(`cost_model`-preferred)-candidate pattern so the @@ -752,7 +752,7 @@ async fn sql_count_distinct_with_epsilon_binds_hll_rse_over_named_column() { /// An exact workload binds zero sketches: `SUM(bytes) GROUP BY service` at /// `AccuracyTarget::Exact` still gets its mergeable exact accumulator, and -/// `AVG(bytes)` (non-mergeable) stays a whole logical subtree untouched. SQL +/// `AVG(bytes)` (non-mergeable) stays a whole logical sub-DAG untouched. SQL /// counterpart of `promql_to_post_asap.rs`'s /// `promql_exact_workload_binds_accumulators_not_sketches`. #[tokio::test] diff --git a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs index d3d73121b..ed961cc85 100644 --- a/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs +++ b/crates/integration-tests/tests/summary_maintenance_lifecycle_e2e.rs @@ -174,8 +174,8 @@ fn promql_dashboard_materializes_continuous_summary_with_explained_rejections() alternative["lifecycle"]["kind"] == "shared" && alternative["rejection"] == "unsupported_by_runtime" })); - assert!(exported["graph"]["nodes"].as_array().is_some()); - let summary_node = exported["graph"]["nodes"] + assert!(exported["dag"]["nodes"].as_array().is_some()); + let summary_node = exported["dag"]["nodes"] .as_array() .unwrap() .iter() @@ -448,7 +448,7 @@ fn quantile_workload(query: &str) -> PlanningWorkload { fn lifecycle_timed_dag( query: &str, lifecycle: &SummaryMaintenanceLifecycle, -) -> (asap_types::post_asap::PostAsapDag, Vec) { +) -> (asap_types::post_asap::PostAsapDAG, Vec) { use asap_aware_mapping::enumerate_summary_maintenance_lifecycles; let workload = quantile_workload(query); let mut lowered = lower_promql_workload(&workload, 0).unwrap().remove(0); @@ -489,7 +489,7 @@ fn lifecycle_timed_dag( /// Compile inputs for a timed DAG: its raw source, available at either phase. fn raw_inputs( - dag: &asap_types::post_asap::PostAsapDag, + dag: &asap_types::post_asap::PostAsapDAG, ) -> std::collections::BTreeMap { let raw = dag .nodes @@ -965,7 +965,7 @@ fn grouped_rate_sum_placement_is_a_lifecycle_choice() { /// The lifecycle-timed DAG Planner selects for `query` with upfront series /// typing, and whether it keeps an ingestion-time Binary. -fn typed_selection(query: &str) -> (asap_types::post_asap::PostAsapDag, bool) { +fn typed_selection(query: &str) -> (asap_types::post_asap::PostAsapDAG, bool) { use asap_types::post_asap::{ExecutionTiming, PostAsapOperatorPayload}; let workload = quantile_workload(query); let lowered = asap_types::pre_asap::schema::with_promql_series_identity( @@ -982,10 +982,10 @@ fn typed_selection(query: &str) -> (asap_types::post_asap::PostAsapDag, bool) { (dag, ingestion_binary) } -/// Execute a timed DAG's precompute and query graphs over `samples` +/// Execute a timed DAG's precompute and query DAGs over `samples` /// (`(metric, job, seconds, value)`) at 300s; returns the root's values. fn execute_timed( - dag: &asap_types::post_asap::PostAsapDag, + dag: &asap_types::post_asap::PostAsapDAG, samples: &[(&str, &str, i64, f64)], ) -> Vec { use asap_physical_operators::{ @@ -1063,7 +1063,7 @@ fn execute_timed( &frontier, ) .unwrap(); - let raw_sources = |plan: &asap_physical_operators::physical_planner::CompiledPhysicalDag| { + let raw_sources = |plan: &asap_physical_operators::physical_planner::CompiledPhysicalDAG| { plan.input_contracts() .filter_map(|(id, _)| raw.get(&id).map(|(schema, name)| (id, batch(schema, name)))) .collect::>() diff --git a/crates/planner/tests/summary_sharing.rs b/crates/planner/tests/summary_sharing.rs index 829c38a9c..069bda8ff 100644 --- a/crates/planner/tests/summary_sharing.rs +++ b/crates/planner/tests/summary_sharing.rs @@ -21,7 +21,7 @@ use asap_frontend_promql::lower_promql_workload; use asap_frontend_sql::SqlCatalog; use asap_planner::{e2e_plan, FrontendInput, UserInput}; use asap_types::post_asap::{ - share_common_summary_subtrees, AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, + share_common_summary_sub_dags, AccuracyError, BoundExpr, CompositionOperator, ErrorMetric, ProbabilityExpr, ResultGuarantee, SketchQuery, }; use asap_types::post_asap::{ @@ -542,7 +542,7 @@ fn certified_frequency_readouts_share_one_univmon_state() { }) .collect(); let mut states: Vec> = Vec::new(); - for (_, root) in share_common_summary_subtrees(assembled) { + for (_, root) in share_common_summary_sub_dags(assembled) { assert!(root.guarantee.is_some(), "{:?}", root.expr); let SummaryExpr::SummaryEstimate { summary_input, .. } = &root.expr else { panic!("summary readout: {:?}", root.expr); diff --git a/crates/types/Cargo.toml b/crates/types/Cargo.toml index 6b3536575..8572854a9 100644 --- a/crates/types/Cargo.toml +++ b/crates/types/Cargo.toml @@ -10,7 +10,7 @@ edition = "2021" [dependencies] # "rc" — QueryExpr's child fields are Rc> (issue #212, #222: # shared sub-expressions), and Rc's Serialize/Deserialize impls live behind -# this feature flag. dag_export.rs / DagNode already flatten the tree to a +# this feature flag. dag_export.rs / DAGNode already flatten the DAG to a # node+edge list for JSON export, so this does not change wire format — a # shared Rc still (de)serializes as an ordinary inline value, once per # reference, exactly like the old Box. diff --git a/crates/types/src/cost.rs b/crates/types/src/cost.rs index 66f4febc1..660d3bb73 100644 --- a/crates/types/src/cost.rs +++ b/crates/types/src/cost.rs @@ -420,8 +420,8 @@ where /// Whole-selected-workload cost/benefit — one query's (or one workload /// batch's) aggregate baseline, selected, and benefit, built from /// [`sum_workload_costs`] over that scope's own per-decision node -/// annotations. See [`crate::dag_export::NamedGraph::workload_cost`] / -/// [`crate::dag_export::WorkloadGraph::workload_cost`]. +/// annotations. See [`crate::dag_export::NamedDAG::workload_cost`] / +/// [`crate::dag_export::WorkloadDAG::workload_cost`]. #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] pub struct WorkloadCostSummary { pub baseline_cost: CostAnnotation, diff --git a/crates/types/src/dag_export.rs b/crates/types/src/dag_export.rs index 0fe6977f9..e5689de3c 100644 --- a/crates/types/src/dag_export.rs +++ b/crates/types/src/dag_export.rs @@ -1,44 +1,44 @@ -//! Export the pre-ASAP [`QueryExpr`] tree as a generic node/edge graph, for tools +//! Export the pre-ASAP [`QueryExpr`] DAG as a generic node/edge DAG, for tools //! that need to render or diff the IR (the `dag_export` example + the //! `tools/dag-viewer` viewer — see issue #133) rather than walk it in Rust. //! -//! `QueryExpr` already derives `Serialize`, but as a Rust-shaped tagged tree +//! `QueryExpr` already derives `Serialize`, but as a Rust-shaped tagged DAG //! (`Rc` children nested inside each variant's own field). This module //! flattens that into an explicit node list + child-id edges — the shape a -//! generic graph renderer wants — and additionally tags each node with +//! generic DAG renderer wants — and additionally tags each node with //! [`structural_hash`](crate::pre_asap::cse::structural_hash), so a caller -//! with several exported queries can spot identical subtrees (a +//! with several exported queries can spot identical sub-DAGs (a //! shared `Scan`, a repeated `Aggregate` shape, …) by comparing hashes //! rather than re-implementing `QueryExpr: PartialEq` structural comparison //! client-side. //! //! This is literally the same hashing -//! [`share_common_subtrees`](crate::pre_asap::cse::share_common_subtrees) +//! [`share_common_sub_dags`](crate::pre_asap::cse::share_common_sub_dags) //! uses to bucket candidates in its `InternTable` (issue #223 stage 3) — not -//! a parallel reimplementation. `tools/dag-viewer`'s "shared subtree" +//! a parallel reimplementation. `tools/dag-viewer`'s "shared sub-DAG" //! highlighting is still a *proxy* for real CSE, though: a hash match here //! only means two nodes are legal `InternTable` bucket-mates (same coarse //! hash), the same candidate-narrowing step `structural_hash` performs -//! inside `InternTable::intern` — it does not mean `share_common_subtrees` +//! inside `InternTable::intern` — it does not mean `share_common_sub_dags` //! actually ran on this data and merged them onto one `Rc` (that also //! requires the `PartialEq` check `InternTable::intern` performs, and the //! `Schema::has_unique_key` legality gate, neither of which this export //! step evaluates). See `tools/dag-viewer/README.md` for the up-to-date //! caveat. //! -//! ## `DagNode::notes` — a layering seam, not a feature this module implements +//! ## `DAGNode::notes` — a layering seam, not a feature this module implements //! -//! [`DagNode`] also carries `notes: Vec<`[`DagNote`]`>`, always empty coming +//! [`DAGNode`] also carries `notes: Vec<`[`DAGNote`]`>`, always empty coming //! out of [`export`]. It exists so a *higher* layer — one that depends on -//! `asap_types`, never the reverse — can annotate an already-exported graph +//! `asap_types`, never the reverse — can annotate an already-exported DAG //! after the fact without this module needing to know anything about that //! layer's concepts. Concretely: `asap-aware-mapping`'s `explanation` module //! (issue #257) computes `structural_hash` over the same `QueryExpr` -//! subtrees this module does (via the identical function). The devtools +//! sub-DAGs this module does (via the identical function). The devtools //! exporter uses that hash to narrow candidates, then compares -//! `ReplacementExplanation::target` with [`DagNode::source_expr`] for a -//! collision-safe match before pushing a [`DagNote`] onto the node. -//! `asap_types` itself never constructs a `DagNote` — see [`DagNode::notes`] +//! `ReplacementExplanation::target` with [`DAGNode::source_expr`] for a +//! collision-safe match before pushing a [`DAGNote`] onto the node. +//! `asap_types` itself never constructs a `DAGNote` — see [`DAGNode::notes`] //! for the layering rule this keeps. use std::collections::HashMap; @@ -55,11 +55,11 @@ use crate::pre_asap::query_expr::{QueryExpr, Source}; /// (predicates, aggregate funcs, schema, sort keys, …) — everything except /// its children, which live in `children` instead. #[derive(Debug, Clone, Serialize)] -pub struct DagNode { +pub struct DAGNode { pub id: u32, /// The `QueryExpr` variant name (e.g. `"Aggregate"`). pub kind: &'static str, - /// Short human-readable summary for a node's collapsed on-graph label. + /// Short human-readable summary for a node's collapsed on-DAG label. pub label: String, pub detail: serde_json::Value, /// Output schema carried by every exported node. Edge renderers use the @@ -75,16 +75,16 @@ pub struct DagNode { #[serde(skip_serializing_if = "Option::is_none")] pub workload_node_id: Option, /// [`structural_hash`](crate::pre_asap::cse::structural_hash) of the - /// subtree rooted at this node — the exact same function `cse`'s + /// sub-DAG rooted at this node — the exact same function `cse`'s /// `InternTable` uses to bucket CSE candidates, so two nodes hash /// equally here iff they would land in the same `InternTable` bucket. /// See the module doc for what a hash match here does and doesn't /// guarantee. /// /// `None` for the same reason `source_expr` is `None` — a post-ASAP- - /// originated node in an [`export_post_asap`] merged graph has no + /// originated node in an [`export_post_asap`] merged DAG has no /// `QueryExpr` to hash. Omitted from JSON entirely (rather than, say, - /// serialized as `0`) so a consumer's shared-subtree-by-hash pass can + /// serialized as `0`) so a consumer's shared-sub-DAG-by-hash pass can /// tell "no hash" apart from a real hash that happens to collide with a /// placeholder — `0` is a legal `structural_hash` output, not a safe /// sentinel. @@ -97,7 +97,7 @@ pub struct DagNode { /// /// `None` for a node with no corresponding pre-ASAP `QueryExpr` at all — /// only possible for a post-ASAP-originated node inside a merged - /// [`export_post_asap`] graph (a `SummaryAgg`/`SummaryJoin`/… node has no + /// [`export_post_asap`] DAG (a `SummaryAgg`/`SummaryJoin`/… node has no /// single `QueryExpr` it corresponds to). Every node [`export`] itself /// produces is pre-ASAP by construction and always carries `Some`. #[serde(skip)] @@ -112,26 +112,26 @@ pub struct DagNode { /// (it has no notion of a "replacement" at all — see the module doc's /// layering note); a higher layer that does (`asap-aware-mapping`, via /// the `dag_export` devtools binary) fills it in after the fact by - /// matching [`DagNode::hash`] and confirming structural equality. Empty + /// matching [`DAGNode::hash`] and confirming structural equality. Empty /// by default, so every existing [`export`] caller and test is unaffected. #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub notes: Vec, + pub notes: Vec, /// Explicit workload-level decision that produced or carried this node. - /// Present only in `post_graph`; consumers must read this rather than - /// infer strategy provenance from labels, hashes, or graph similarity. + /// Present only in `post_dag`; consumers must read this rather than + /// infer strategy provenance from labels, hashes, or DAG similarity. #[serde(skip_serializing_if = "Option::is_none")] - pub decision: Option, + pub decision: Option, } -/// One reporting-layer annotation attached to a [`DagNode`] by a higher -/// layer than `asap_types` — see [`DagNode::notes`]. `asap_types` defines +/// One reporting-layer annotation attached to a [`DAGNode`] by a higher +/// layer than `asap_types` — see [`DAGNode::notes`]. `asap_types` defines /// this shape (so the field has a concrete, serializable type) but never /// constructs one: `asap_types` is a lower crate that `asap-aware-mapping` /// depends on, never the reverse, so this type is deliberately generic and /// crate-agnostic rather than naming anything from that higher layer (e.g. /// its `ExplanationKind`/`ReplacementExplanation`). #[derive(Debug, Clone, Serialize)] -pub struct DagNote { +pub struct DAGNote { /// A short tag for the kind of annotation this is (e.g. a /// `Debug`-formatted `asap_aware_mapping::ExplanationKind`) — opaque to /// `asap_types`, meant for a renderer to group or color by. @@ -144,7 +144,7 @@ pub struct DagNote { /// Self-contained explanation of a winning workload-level post-ASAP /// decision, serialized directly on every node it produced or carried. #[derive(Debug, Clone, Serialize)] -pub struct DagDecision { +pub struct DAGDecision { pub id: u32, pub strategy: String, pub rationale: String, @@ -164,19 +164,19 @@ pub struct DagDecision { #[serde(skip_serializing_if = "Option::is_none")] pub selected_cost: Option, /// `baseline_cost.value - selected_cost.value` under `baseline_cost`'s - /// own baseline — for a winning `SharedSubtreeStrategy`/`CseShare` + /// own baseline — for a winning `SharedSubDAGStrategy`/`CseShare` /// decision this *is* "avoided recomputation for a shared sub-DAG" (one /// of `dag_export`'s issue #286 granularity items): the baseline is - /// exactly the cost of recomputing this subtree independently at every + /// exactly the cost of recomputing this sub-DAG independently at every /// consumer, so the benefit is exactly what sharing avoided. #[serde(skip_serializing_if = "Option::is_none")] pub benefit: Option, } -/// A cost/benefit annotation attributed to one specific graph edge (`from` -/// -> `to`, in [`DagNode::children`]'s direction) rather than to a node — +/// A cost/benefit annotation attributed to one specific DAG edge (`from` +/// -> `to`, in [`DAGNode::children`]'s direction) rather than to a node — /// issue #286's "edge cost only when genuinely attributable to the edge" -/// granularity item. Graph structure alone cannot determine transfer, +/// granularity item. DAG structure alone cannot determine transfer, /// materialization, or read cost. A higher layer may attach this annotation /// only when physical evidence attributes cost to this exact edge; this /// module never derives one from structural node counts. @@ -187,15 +187,15 @@ pub struct EdgeCostAnnotation { pub cost: CostAnnotation, } -/// One query's exported graph. `nodes[root as usize]` is the tree's root. +/// One query's exported DAG. `nodes[root as usize]` is the DAG's root. #[derive(Debug, Clone, Serialize)] -pub struct DagGraph { - pub nodes: Vec, +pub struct ExportDAG { + pub nodes: Vec, pub root: u32, /// See [`EdgeCostAnnotation`]. Always empty unless a higher layer - /// explicitly populated it (same layering rule as [`DagNode::notes`]); + /// explicitly populated it (same layering rule as [`DAGNode::notes`]); /// omitted from JSON entirely when empty, so every existing producer of - /// [`DagGraph`] (every call to [`export`]/[`export_summary`]) is + /// [`ExportDAG`] (every call to [`export`]/[`export_summary`]) is /// unaffected. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub edge_annotations: Vec, @@ -203,57 +203,57 @@ pub struct DagGraph { /// A single named query within a multi-query export. #[derive(Debug, Clone, Serialize)] -pub struct NamedGraph { +pub struct NamedDAG { pub name: String, - /// The original query text (SQL or PromQL) this graph was lowered from, - /// for display alongside the graph — not used by `export` itself, since + /// The original query text (SQL or PromQL) this DAG was lowered from, + /// for display alongside the DAG — not used by `export` itself, since /// that only sees the already-lowered `QueryExpr`. Optional because not - /// every producer of a `NamedGraph` has the source text on hand. + /// every producer of a `NamedDAG` has the source text on hand. #[serde(skip_serializing_if = "Option::is_none")] pub source: Option, - pub graph: DagGraph, + pub dag: ExportDAG, /// Concrete post-ASAP replacement sites discovered for this query — see /// [`TargetReplacement`]. Always empty coming out of anything in this - /// module (same layering rule as [`DagNode::notes`]: `asap_types` never + /// module (same layering rule as [`DAGNode::notes`]: `asap_types` never /// runs `asap-aware-mapping`'s search itself); a higher layer populates /// this after the fact, e.g. the `dag_export` devtools binary's /// `--post-asap` flag. Omitted from the JSON entirely when empty, so - /// every existing producer/consumer of `NamedGraph` (in particular every + /// every existing producer/consumer of `NamedDAG` (in particular every /// invocation of `dag_export` without `--post-asap`) keeps emitting and /// parsing exactly the same shape it always has. #[serde(default, skip_serializing_if = "Vec::is_empty")] pub replacements: Vec, - /// One merged "whole query, but post-ASAP" graph — see + /// One merged "whole query, but post-ASAP" DAG — see /// [`export_post_asap`] for how a higher layer builds this. Unlike /// [`TargetReplacement::before`]/`::after` (small, self-contained /// before/after pairs, one per independently-discovered replacement - /// site), this is a single flattened [`DagGraph`] spanning the whole + /// site), this is a single flattened [`ExportDAG`] spanning the whole /// query: every node that has no winning replacement renders as an - /// ordinary pre-ASAP [`DagNode`] (same shape [`export`] itself + /// ordinary pre-ASAP [`DAGNode`] (same shape [`export`] itself /// produces), and every node that does splices in its winning - /// candidate's shape instead — a rewritten [`QueryExpr`] subtree, or a - /// bound `SummaryNode` subtree, rendered inline in the very same node + /// candidate's shape instead — a rewritten [`QueryExpr`] sub-DAG, or a + /// bound `SummaryNode` sub-DAG, rendered inline in the very same node /// list. `None` unless a higher layer explicitly built one (e.g. the /// `dag_export` devtools binary's `--post-asap` flag); omitted from the /// JSON entirely when absent, so every existing producer/consumer of - /// `NamedGraph` is unaffected. + /// `NamedDAG` is unaffected. #[serde(default, skip_serializing_if = "Option::is_none")] - pub post_graph: Option, + pub post_dag: Option, /// This query's own selected-workload cost/benefit — one of issue /// #286's granularity items. Built by summing *this query's own* - /// `post_graph` decision-node cost annotations, deduplicated by + /// `post_dag` decision-node cost annotations, deduplicated by /// `decision.id` **within this one query only** (a decision spanning /// several nodes in this query's own replacement region is still /// counted once here). `None` unless a higher layer built one (same - /// `--post-asap`-gated pattern as `post_graph`); omitted from JSON when + /// `--post-asap`-gated pattern as `post_dag`); omitted from JSON when /// absent. /// /// This does **not** dedupe across queries: a target shared by two /// queries (e.g. a common `Scan` after workload-wide CSE) is counted /// once in *each* query's own `workload_cost` — summing several - /// `NamedGraph.workload_cost` values by hand double-counts any decision + /// `NamedDAG.workload_cost` values by hand double-counts any decision /// shared between them. For a cross-query total that dedupes correctly, - /// use [`WorkloadGraph::workload_cost`] instead, which is built + /// use [`WorkloadDAG::workload_cost`] instead, which is built /// specifically to cover every query in one pass. #[serde(default, skip_serializing_if = "Option::is_none")] pub workload_cost: Option, @@ -266,12 +266,12 @@ pub struct NamedGraph { } /// A batch of named queries — the shape the viewer's multi-query / compare -/// mode reads (each query starts its own `DagGraph`; shared-subtree -/// highlighting is done by the viewer, matching `DagNode::hash` across +/// mode reads (each query starts its own `ExportDAG`; shared-sub-DAG +/// highlighting is done by the viewer, matching `DAGNode::hash` across /// queries). #[derive(Debug, Clone, Serialize)] -pub struct WorkloadGraph { - pub queries: Vec, +pub struct WorkloadDAG { + pub queries: Vec, /// The selected multi-query workload's own cost/benefit, deduplicated /// across every query in `queries` (not just within one) — the /// "Selecting ... multiple queries ... display correct Pre/Post-ASAP @@ -290,7 +290,7 @@ pub struct WorkloadGraph { // generic, crate-agnostic "one replacement site, before and after" shape a // higher layer (`asap-aware-mapping`, via the `dag_export` devtools binary's // `--post-asap` flag) populates after running its own search — the exact -// same layering rule [`DagNode::notes`]'s doc above already states: this +// same layering rule [`DAGNode::notes`]'s doc above already states: this // module never runs `asap_aware_mapping::replacement::search_workload_with` // itself, never picks a "winning" candidate, and has no opinion on what a // `ReplacementProvenance` or a cost model even is. It only defines shapes @@ -298,32 +298,32 @@ pub struct WorkloadGraph { // `tools/dag-viewer` to render without needing to know anything about // `asap-aware-mapping`'s own vocabulary. // -// A single whole-query "post-ASAP tree" isn't attempted here, and isn't +// A single whole-query "post-ASAP DAG" isn't attempted here, and isn't // representable in the current type system either: `SummaryExpr` has no // variant letting a `SummaryNode` be embedded back inside a plain // `QueryExpr`'s child slot (`QueryExpr`'s own children are always // `Rc`, never `Rc`), so there is no way to splice a -// post-ASAP binding back into its original pre-ASAP tree in place. Inventing +// post-ASAP binding back into its original pre-ASAP DAG in place. Inventing // a bridge type for that is a real `asap_types`/`asap-aware-mapping` IR // design decision, well beyond what a devtools visualization export should // decide unilaterally. Instead, each independently-discovered replacement // target gets its own small, self-contained `before`/`after` pair — the -// target's own pre-ASAP subtree, and either the winning `SummaryNode` or the +// target's own pre-ASAP sub-DAG, and either the winning `SummaryNode` or the // winning rewritten `QueryExpr`, both of which *are* fully representable // today via [`export`]/[`export_summary`] as-is. /// One flattened post-ASAP node — the [`SummaryExpr`] analogue of -/// [`DagNode`]. `detail` holds this node's own scalar fields (the summarized +/// [`DAGNode`]. `detail` holds this node's own scalar fields (the summarized /// column, the summary family, grouping strategy, sketch-query kind, …) — /// everything except its `SummaryNode` children, which live in `children` /// instead. /// -/// Unlike [`DagNode`], this carries no `hash`/`source_expr` pair: nothing in -/// this module ever needs to re-identify a particular `SummaryDagNode` the -/// way `DagNode::hash` lets a higher layer re-identify a pre-ASAP node (a +/// Unlike [`DAGNode`], this carries no `hash`/`source_expr` pair: nothing in +/// this module ever needs to re-identify a particular `SummaryDAGNode` the +/// way `DAGNode::hash` lets a higher layer re-identify a pre-ASAP node (a /// `SummaryNode` is always freshly exported for exactly one /// [`TargetReplacementAfter::Summary`] site, never matched back against a -/// separately-exported graph the way pre-ASAP notes are). +/// separately-exported DAG the way pre-ASAP notes are). /// /// Several of `SummaryExpr`'s own fields (`SummaryFamilyType`, /// `GroupingStrategy`, `SketchQuery`) derive neither `Serialize` nor @@ -335,15 +335,15 @@ pub struct WorkloadGraph { /// reporting concern), this module renders those particular fields into /// `detail` via their `Debug` formatting instead — human-readable, and /// sufficient for the display purpose `detail` exists for on every other -/// node in this file (see [`DagNode::detail`]'s own doc), at the cost of +/// node in this file (see [`DAGNode::detail`]'s own doc), at the cost of /// those particular fields being opaque strings rather than structured JSON -/// on the `SummaryDagNode` side of the export. +/// on the `SummaryDAGNode` side of the export. #[derive(Debug, Clone, Serialize)] -pub struct SummaryDagNode { +pub struct SummaryDAGNode { pub id: u32, /// The `SummaryExpr` variant name (e.g. `"SummaryAgg"`). pub kind: &'static str, - /// Short human-readable summary for a node's collapsed on-graph label. + /// Short human-readable summary for a node's collapsed on-DAG label. pub label: String, pub detail: serde_json::Value, /// Child node ids, in the variant's field order (e.g. `SummaryJoin` is @@ -363,11 +363,11 @@ pub struct SummaryDagNode { /// target (issue #172) — `asap_aware_mapping::replacement::RejectedCandidate` /// re-shaped into this crate's own crate-agnostic vocabulary, the same /// layering rule as [`TargetReplacement`]. Carried on -/// [`NamedGraph::rejections`] so a renderer can explain *why* a target kept +/// [`NamedDAG::rejections`] so a renderer can explain *why* a target kept /// its raw/pre-ASAP form, not only what won elsewhere. #[derive(Debug, Clone, Serialize)] pub struct TargetRejection { - /// Id of the [`DagNode`] in this query's own `graph.nodes` the refused + /// Id of the [`DAGNode`] in this query's own `DAG.nodes` the refused /// candidate targeted. pub target_pre_id: u32, /// Which strategy considered the candidate. @@ -378,38 +378,38 @@ pub struct TargetRejection { pub error: AccuracyError, } -/// One post-ASAP `SummaryNode` tree, flattened the same way [`DagGraph`] -/// flattens a pre-ASAP `QueryExpr` tree. +/// One post-ASAP `SummaryNode` DAG, flattened the same way [`ExportDAG`] +/// flattens a pre-ASAP `QueryExpr` DAG. #[derive(Debug, Clone, Serialize)] -pub struct SummaryDagGraph { - pub nodes: Vec, +pub struct SummaryDAG { + pub nodes: Vec, pub root: u32, } /// Flatten a [`SummaryNode`] the same way [`export`] flattens a `QueryExpr` -/// — post-order, one [`SummaryDagNode`] per [`SummaryExpr`] variant, no +/// — post-order, one [`SummaryDAGNode`] per [`SummaryExpr`] variant, no /// memoization of repeated `Rc` references (a shared /// sub-expression reachable through two parents is flattened twice, into two -/// separate node entries — the same "this is a flattened tree view, not a -/// pointer-identity-preserving graph" behavior [`build`] already has for +/// separate node entries — the same "this is a flattened DAG view, not a +/// pointer-identity-preserving DAG" behavior [`build`] already has for /// `QueryExpr`). /// -/// A `KeepPreAsap(inner)` leaf embeds the *whole* pre-ASAP subtree beneath it -/// as a nested [`DagGraph`] (via [`export(inner)`](export)) inside its own -/// `detail` field (`{"pre_asap_subgraph": }`) rather than trying to -/// flatten it into this same node list — [`DagNode`] and [`SummaryDagNode`] +/// A `KeepPreAsap(inner)` leaf embeds the *whole* pre-ASAP sub-DAG beneath it +/// as a nested [`ExportDAG`] (via [`export(inner)`](export)) inside its own +/// `detail` field (`{"pre_asap_sub_dag": }`) rather than trying to +/// flatten it into this same node list — [`DAGNode`] and [`SummaryDAGNode`] /// are different types with different id spaces, so mixing them into one /// `Vec` isn't type-safe; nesting is. `label` for a `KeepPreAsap` node is -/// `format!("KeepPreAsap({kind})")`, where `kind` is the inner subtree's own -/// top-level `DagNode::kind`. -pub fn export_summary(node: &SummaryNode) -> SummaryDagGraph { +/// `format!("KeepPreAsap({kind})")`, where `kind` is the inner sub-DAG's own +/// top-level `DAGNode::kind`. +pub fn export_summary(node: &SummaryNode) -> SummaryDAG { let mut nodes = Vec::new(); let root = build_summary(node, &mut nodes); - SummaryDagGraph { nodes, root } + SummaryDAG { nodes, root } } fn push_summary_node( - nodes: &mut Vec, + nodes: &mut Vec, kind: &'static str, label: String, detail: serde_json::Value, @@ -417,7 +417,7 @@ fn push_summary_node( guarantee: Option, ) -> u32 { let id = nodes.len() as u32; - nodes.push(SummaryDagNode { + nodes.push(SummaryDAGNode { id, kind, label, @@ -430,7 +430,7 @@ fn push_summary_node( /// A short, human-readable label for a [`crate::post_asap::SummaryFamilyType`] /// (e.g. `"Sketch(Kll)"`, `"ExactAggregate(Sum)"`) — for -/// [`SummaryDagNode::label`] text on a `SummaryAgg`/`SummaryJoin` node. Not +/// [`SummaryDAGNode::label`] text on a `SummaryAgg`/`SummaryJoin` node. Not /// exhaustive prose (mirrors `asap_aware_mapping::replacement::describe_intent`'s /// own "this is a label, not a decision" stance) — every variant is covered, /// but via `Debug` for the inner kind rather than hand-written prose per @@ -448,13 +448,13 @@ fn family_label(family: &crate::post_asap::SummaryFamilyType) -> String { } /// `(kind, label, detail)` for every [`SummaryExpr`] variant *except* -/// [`SummaryExpr::KeepPreAsap`] — that variant has no `SummaryDagNode`/ -/// `DagNode` of its own (see [`build_summary`]/[`build_summary_hybrid`], its +/// [`SummaryExpr::KeepPreAsap`] — that variant has no `SummaryDAGNode`/ +/// `DAGNode` of its own (see [`build_summary`]/[`build_summary_hybrid`], its /// only two callers, both of which special-case it before ever reaching /// this function). Factored out so [`build_summary`] (nests a `KeepPreAsap` -/// leaf's pre-ASAP subtree as its own [`SummaryDagGraph`]) and -/// [`build_summary_hybrid`] (splices that same subtree directly into a -/// shared [`DagGraph`] node list — see [`export_post_asap`]) can't drift +/// leaf's pre-ASAP sub-DAG as its own [`SummaryDAG`]) and +/// [`build_summary_hybrid`] (splices that same sub-DAG directly into a +/// shared [`ExportDAG`] node list — see [`export_post_asap`]) can't drift /// apart on how every *other* variant's own shape is described, since /// nothing about that description differs between the two. macro_rules! define_summary_kind_tags { @@ -584,16 +584,16 @@ fn summary_children(expr: &SummaryExpr) -> Vec<&Rc> { } } -/// Recursively flatten `node`, appending [`SummaryDagNode`]s to `nodes` in +/// Recursively flatten `node`, appending [`SummaryDAGNode`]s to `nodes` in /// post-order (children pushed before their parent), and return the pushed /// root's id. Exhaustive over every [`SummaryExpr`] variant, matching this /// file's own exhaustive style for `QueryExpr` in [`build`]. -fn build_summary(node: &SummaryNode, nodes: &mut Vec) -> u32 { +fn build_summary(node: &SummaryNode, nodes: &mut Vec) -> u32 { if let SummaryExpr::KeepPreAsap(inner) = &node.expr { - let pre_asap_subgraph = export(inner); - let inner_kind = pre_asap_subgraph.nodes[pre_asap_subgraph.root as usize].kind; + let pre_asap_sub_dag = export(inner); + let inner_kind = pre_asap_sub_dag.nodes[pre_asap_sub_dag.root as usize].kind; let label = format!("KeepPreAsap({inner_kind})"); - let detail = serde_json::json!({ "pre_asap_subgraph": pre_asap_subgraph }); + let detail = serde_json::json!({ "pre_asap_sub_dag": pre_asap_sub_dag }); return push_summary_node( nodes, "KeepPreAsap", @@ -615,21 +615,21 @@ fn build_summary(node: &SummaryNode, nodes: &mut Vec) -> u32 { /// running `asap_aware_mapping::replacement::search_workload_with` + /// `CandidateLogicalASAPDAGs::cost_sorted` and picking the best-ranked candidate for one /// `TargetSubDAGCandidates` — `asap_types` never runs that search itself (same layering -/// rule as [`DagNote`]: this crate defines the shape, a higher crate +/// rule as [`DAGNote`]: this crate defines the shape, a higher crate /// populates it). #[derive(Debug, Clone, Serialize)] pub struct TargetReplacement { /// Stable id of this workload-level winning decision. Nodes in - /// [`NamedGraph::post_graph`] produced by this decision carry the same id, + /// [`NamedDAG::post_dag`] produced by this decision carry the same id, /// so renderers can explain a clicked post-ASAP node without guessing by - /// label, hash, or graph shape. + /// label, hash, or DAG shape. pub decision_id: u32, - /// Id of the [`DagNode`] (in this query's own `graph.nodes`, i.e. the - /// [`NamedGraph`] this `TargetReplacement` is attached to) this - /// replacement's `before` subtree is rooted at. + /// Id of the [`DAGNode`] (in this query's own `DAG.nodes`, i.e. the + /// [`NamedDAG`] this `TargetReplacement` is attached to) this + /// replacement's `before` sub-DAG is rooted at. pub target_pre_id: u32, /// Human label for which strategy proposed the winning candidate — - /// e.g. `"Sketch"` / `"HydraGrouping"` / `"SharedSubtree"` / + /// e.g. `"Sketch"` / `"HydraGrouping"` / `"SharedSubDAG"` / /// `"AvgToSumCountRewrite"` / `"Rollup"`. The higher layer derives this from /// `ReplacementProvenance` plus which strategy's shape actually /// produced the winning candidate; `asap_types` has no opinion on the @@ -648,9 +648,9 @@ pub struct TargetReplacement { /// doesn't estimate a numeric cost for this candidate shape (see that /// field's own doc upstream). pub cost: f64, - /// The target's own pre-ASAP subtree, before replacement — literally + /// The target's own pre-ASAP sub-DAG, before replacement — literally /// `export(target)` for the `TargetSubDAGCandidates`'s own `target`, reused as-is. - pub before: DagGraph, + pub before: ExportDAG, pub after: TargetReplacementAfter, /// Structured baseline/selected/benefit cost annotations for this one /// replacement region — issue #286's "replacement-region baseline @@ -658,7 +658,7 @@ pub struct TargetReplacement { /// consistent with `cost` above: `selected_cost.value == Some(cost)` /// whenever `cost` is finite, `None`/`Unavailable` whenever it is /// `NaN`. Baseline and selected values require complete, scope-matched - /// physical evidence; neither is inferred from logical graph structure. + /// physical evidence; neither is inferred from logical DAG structure. #[serde(skip_serializing_if = "Option::is_none")] pub baseline_cost: Option, #[serde(skip_serializing_if = "Option::is_none")] @@ -671,25 +671,25 @@ pub struct TargetReplacement { /// or a still-pre-ASAP-shaped structural rewrite, mirroring /// `asap_aware_mapping::replacement::Replacement`'s own two variants. /// -/// Serializes as `{"kind": "Summary"|"Rewrite", "graph": {...}}` (serde's +/// Serializes as `{"kind": "Summary"|"Rewrite", "DAG": {...}}` (serde's /// adjacently-tagged representation for a `#[serde(tag = "kind", content = -/// "graph")]` enum) — this exact shape is a cross-team contract with +/// "DAG")]` enum) — this exact shape is a cross-team contract with /// `tools/dag-viewer`'s fixture data, so it isn't incidental: changing it /// needs coordinating with that side, not just a local refactor here. #[derive(Debug, Clone, Serialize)] -#[serde(tag = "kind", content = "graph")] +#[serde(tag = "kind", content = "dag")] pub enum TargetReplacementAfter { /// A `Replacement::Summary` candidate — a genuine post-ASAP binding. - Summary(SummaryDagGraph), + Summary(SummaryDAG), /// A `Replacement::Rewrite` candidate — still pre-ASAP shaped (CSE /// share/recompute, `AvgToSumOverCountStrategy`, and `RollupStrategy` - /// all produce this kind), so this reuses [`DagGraph`]/[`export`] too, + /// all produce this kind), so this reuses [`ExportDAG`]/[`export`] too, /// not a new type. - Rewrite(DagGraph), + Rewrite(ExportDAG), } -/// Flatten `expr` into a [`DagGraph`]. -pub fn export(expr: &QueryExpr) -> DagGraph { +/// Flatten `expr` into a [`ExportDAG`]. +pub fn export(expr: &QueryExpr) -> ExportDAG { let mut nodes = Vec::new(); // One cache for the whole export — persisted across every `build`/ // `push_node` call, not reset per node, so `structural_hash` memoizes @@ -701,7 +701,7 @@ pub fn export(expr: &QueryExpr) -> DagGraph { // callback regardless (so `export_post_asap` can share this exact // per-variant traversal instead of duplicating it). let root = build(expr, &mut nodes, &mut cache, &mut |_| None); - DagGraph { + ExportDAG { nodes, root, edge_annotations: Vec::new(), @@ -709,42 +709,42 @@ pub fn export(expr: &QueryExpr) -> DagGraph { } /// What a higher layer found for one specific pre-ASAP node when building a -/// merged post-ASAP graph via [`export_post_asap`] — see that function's own +/// merged post-ASAP DAG via [`export_post_asap`] — see that function's own /// doc for the full design. `asap_types` has no opinion on *how* this is /// decided (that's `asap_aware_mapping::replacement::search_workload_with` + /// `CandidateLogicalASAPDAGs::cost_sorted`'s job, a higher layer, exactly the layering rule -/// [`DagNode::notes`] already states); it only defines the shape a decision +/// [`DAGNode::notes`] already states); it only defines the shape a decision /// comes back in. #[derive(Debug, Clone)] pub enum PostAsapSubstitution { /// This exact node has a winning `Replacement::Rewrite` — keep building /// from `.0` instead of the original node. Still pre-ASAP shaped, so - /// [`build`] renders it via the same ordinary `DagNode` path — see + /// [`build`] renders it via the same ordinary `DAGNode` path — see /// [`build`]'s own doc for why `.0`'s own top level is rendered without /// re-querying `find_winner` on it (its descendants still are). Rewrite { replacement: Rc, - decision: DagDecision, + decision: DAGDecision, }, /// This exact node has a winning `Replacement::Summary` — switch to /// rendering `.0`'s bound `SummaryNode` shape from here down, via /// [`build_summary_hybrid`]. Summary { replacement: Rc, - decision: DagDecision, + decision: DAGDecision, }, } -/// Build one merged "whole query, but post-ASAP" [`DagGraph`] by walking +/// Build one merged "whole query, but post-ASAP" [`ExportDAG`] by walking /// `root`'s ordinary pre-ASAP shape and, at every node, asking `find_winner` /// whether *that exact node* has a winning replacement — if so, splicing /// the replacement's own shape in at that position instead, in the very -/// same flattened node list (not a nested sub-graph the way +/// same flattened node list (not a nested sub-DAG the way /// [`TargetReplacement::before`]/`::after` — small, independent, per-site /// before/after pairs — already do; see this file's "Post-ASAP replacement /// export" section doc for why *that* design doesn't attempt a single /// whole-query composite, and why this one can: this is a synthetic -/// id/edge list, the same kind of thing [`DagGraph`] already is for the +/// id/edge list, the same kind of thing [`ExportDAG`] already is for the /// pre-ASAP side, not a real `QueryExpr`/`SummaryNode` value with a type /// system to satisfy). /// @@ -762,7 +762,7 @@ pub enum PostAsapSubstitution { /// substitution's own immediate top level (only on that substitution's /// *descendants*, which get an ordinary fresh call same as any other node). /// This matters for correctness, not just efficiency: -/// `SharedSubtreeStrategy`'s own "build once and share" candidate is +/// `SharedSubDAGStrategy`'s own "build once and share" candidate is /// `Replacement::Rewrite(Rc::clone(target))` — literally the *same* value /// as the target it's a candidate for. Re-querying `find_winner` on that /// candidate's own top level would find the identical group and its @@ -773,14 +773,14 @@ pub enum PostAsapSubstitution { pub fn export_post_asap( root: &QueryExpr, find_winner: &mut dyn FnMut(&QueryExpr) -> Option, -) -> DagGraph { +) -> ExportDAG { let mut nodes = Vec::new(); let mut cache = HashCache::new(); let root_id = build(root, &mut nodes, &mut cache, find_winner); deduplicate_pointer_shared_nodes(nodes, root_id) } -fn deduplicate_pointer_shared_nodes(nodes: Vec, root: u32) -> DagGraph { +fn deduplicate_pointer_shared_nodes(nodes: Vec, root: u32) -> ExportDAG { let mut by_source_ptr = HashMap::::new(); let mut old_to_new = vec![0_u32; nodes.len()]; let mut deduplicated = Vec::with_capacity(nodes.len()); @@ -807,7 +807,7 @@ fn deduplicate_pointer_shared_nodes(nodes: Vec, root: u32) -> DagGraph deduplicated.push(node); } - DagGraph { + ExportDAG { nodes: deduplicated, root: old_to_new[root as usize], edge_annotations: Vec::new(), @@ -868,15 +868,15 @@ define_query_kind_tags! { QueryExpr::BinaryOp { .. } => "BinaryOp", } -/// Push one flattened node for `expr`. `expr` is the *whole* subtree this +/// Push one flattened node for `expr`. `expr` is the *whole* sub-DAG this /// node represents (not just its own fields) — `hash` is /// [`structural_hash(expr)`](structural_hash), the identical function and /// the identical input `InternTable::intern` would hash for this same -/// subtree, so this node's `hash` matches what `cse::share_common_subtrees` +/// sub-DAG, so this node's `hash` matches what `cse::share_common_sub_dags` /// would bucket it under. `kind` is [`kind_tag(expr)`](kind_tag), not a /// caller-supplied argument — see that function's doc for why. fn push_node( - nodes: &mut Vec, + nodes: &mut Vec, expr: &QueryExpr, cache: &mut HashCache, label: String, @@ -885,7 +885,7 @@ fn push_node( ) -> u32 { let id = nodes.len() as u32; let hash = Some(structural_hash(expr, cache)); - nodes.push(DagNode { + nodes.push(DAGNode { id, kind: kind_tag(expr), label, @@ -907,23 +907,23 @@ fn push_node( /// Push one flattened node with no corresponding pre-ASAP `QueryExpr` at /// all — a post-ASAP-originated node inside [`export_post_asap`]'s merged -/// graph (a `SummaryAgg`/`SummaryJoin`/… node, via [`build_summary_hybrid`]). -/// `hash`/`source_expr`-based re-identification (see [`DagNode::hash`]'s own +/// DAG (a `SummaryAgg`/`SummaryJoin`/… node, via [`build_summary_hybrid`]). +/// `hash`/`source_expr`-based re-identification (see [`DAGNode::hash`]'s own /// doc) has no meaning for a node with no `QueryExpr` behind it, so this /// pushes a fixed placeholder hash (`0`) and `source_expr: None` rather than /// inventing a hash over `SummaryExpr` (which, unlike `QueryExpr`, has no -/// [`structural_hash`]-equivalent function at all — see [`SummaryDagNode`]'s +/// [`structural_hash`]-equivalent function at all — see [`SummaryDAGNode`]'s /// own doc on why `SummaryExpr`'s fields don't even derive `Hash`/`PartialEq` /// consistently enough to build one). fn push_summary_originated_node( - nodes: &mut Vec, + nodes: &mut Vec, kind: &'static str, label: String, detail: serde_json::Value, children: Vec, ) -> u32 { let id = nodes.len() as u32; - nodes.push(DagNode { + nodes.push(DAGNode { id, kind, label, @@ -942,18 +942,18 @@ fn push_summary_originated_node( /// The [`build_summary`]/[`build_summary_hybrid`] counterpart of [`build`] /// for a bound [`SummaryNode`] reached while building -/// [`export_post_asap`]'s merged graph: appends into the *same* `nodes: -/// Vec` list `build` itself is filling, instead of a separate -/// [`SummaryDagGraph`]. A `KeepPreAsap(inner)` leaf recurses back into +/// [`export_post_asap`]'s merged DAG: appends into the *same* `nodes: +/// Vec` list `build` itself is filling, instead of a separate +/// [`SummaryDAG`]. A `KeepPreAsap(inner)` leaf recurses back into /// [`build`] on `inner` (the general pre-ASAP entry, `find_winner` included) -/// rather than nesting a `{"pre_asap_subgraph": ...}` blob the way -/// [`build_summary`] does — so the merged graph reads as one seamless graph +/// rather than nesting a `{"pre_asap_sub_dag": ...}` blob the way +/// [`build_summary`] does — so the merged DAG reads as one seamless DAG /// with no dead ends, and so a target reachable underneath a `KeepPreAsap` /// wrapper (a nested aggregate a strategy independently found a /// replacement for, say) still gets spliced in correctly. fn build_summary_hybrid( node: &SummaryNode, - nodes: &mut Vec, + nodes: &mut Vec, cache: &mut HashCache, find_winner: &mut dyn FnMut(&QueryExpr) -> Option, ) -> u32 { @@ -965,9 +965,9 @@ fn build_summary_hybrid( .map(|child| build_summary_hybrid(child, nodes, cache, find_winner)) .collect(); let (kind, label, mut detail) = summary_shape(&node.expr); - // The merged graph's `DagNode` has no dedicated guarantee field (it is + // The merged DAG's `DAGNode` has no dedicated guarantee field (it is // the pre-ASAP node shape); the guarantee rides in `detail` under the - // same key/shape `SummaryDagNode::guarantee` uses, additively. + // same key/shape `SummaryDAGNode::guarantee` uses, additively. if let Some(guarantee) = &node.guarantee { if let (serde_json::Value::Object(map), Ok(value)) = (&mut detail, serde_json::to_value(guarantee)) @@ -1007,7 +1007,7 @@ fn source_label(source: &Source) -> String { /// arm that carries one (`Filter.pred`, `Project.cols`, `Aggregate.having`, …) /// serializes it as opaque `detail` JSON via `Predicate`/`ProjectItem`/ /// `AggIntent`'s own `Serialize` impl, same as before the merge — a scalar -/// subtree was never a separate DAG node, so this doesn't change that. +/// sub-DAG was never a separate DAG node, so this doesn't change that. /// /// `find_winner` is [`export_post_asap`]'s substitution seam, threaded /// through every recursive call (including [`export`]'s own, which always @@ -1021,7 +1021,7 @@ fn source_label(source: &Source) -> String { /// rather than by looping back through this check a second time. fn build( expr: &QueryExpr, - nodes: &mut Vec, + nodes: &mut Vec, cache: &mut HashCache, find_winner: &mut dyn FnMut(&QueryExpr) -> Option, ) -> u32 { @@ -1077,7 +1077,7 @@ fn build( /// `find_winner` query. fn build_no_recheck( expr: &QueryExpr, - nodes: &mut Vec, + nodes: &mut Vec, cache: &mut HashCache, find_winner: &mut dyn FnMut(&QueryExpr) -> Option, ) -> u32 { @@ -1434,11 +1434,11 @@ mod tests { #[test] fn leaf_scan_is_a_single_node() { - let graph = export(&scan("metrics", value_col())); - assert_eq!(graph.nodes.len(), 1); - assert_eq!(graph.root, 0); - assert_eq!(graph.nodes[0].kind, "Scan"); - assert!(graph.nodes[0].children.is_empty()); + let dag = export(&scan("metrics", value_col())); + assert_eq!(dag.nodes.len(), 1); + assert_eq!(dag.root, 0); + assert_eq!(dag.nodes[0].kind, "Scan"); + assert!(dag.nodes[0].children.is_empty()); } /// `export` itself never populates higher-layer annotations. Empty @@ -1446,12 +1446,12 @@ mod tests { /// exports retain their existing shape. #[test] fn export_omits_empty_higher_layer_annotations() { - let graph = export(&scan("metrics", value_col())); - assert!(graph.nodes[0].notes.is_empty()); - assert!(graph.nodes[0].decision.is_none()); - assert!(graph.nodes[0].schema.is_some()); - assert!(graph.edge_annotations.is_empty()); - let json = serde_json::to_string(&graph.nodes[0]).unwrap(); + let dag = export(&scan("metrics", value_col())); + assert!(dag.nodes[0].notes.is_empty()); + assert!(dag.nodes[0].decision.is_none()); + assert!(dag.nodes[0].schema.is_some()); + assert!(dag.edge_annotations.is_empty()); + let json = serde_json::to_string(&dag.nodes[0]).unwrap(); assert!( !json.contains("notes"), "empty `notes` must be skipped, not serialized as `[]`: {json}" @@ -1460,10 +1460,10 @@ mod tests { !json.contains("decision"), "empty `decision` must be skipped, not serialized as `null`: {json}" ); - let graph_json = serde_json::to_string(&graph).unwrap(); + let dag_json = serde_json::to_string(&dag).unwrap(); assert!( - !graph_json.contains("edge_annotations"), - "empty `edge_annotations` must be skipped, not serialized as `[]`: {graph_json}" + !dag_json.contains("edge_annotations"), + "empty `edge_annotations` must be skipped, not serialized as `[]`: {dag_json}" ); } @@ -1487,14 +1487,14 @@ mod tests { children: vec![left_branch, right_branch], discriminator_unique_key: None, }; - let graph = export_post_asap(&root, &mut |_| None); + let dag = export_post_asap(&root, &mut |_| None); assert_eq!( - graph.nodes.iter().filter(|n| n.kind == "Scan").count(), + dag.nodes.iter().filter(|n| n.kind == "Scan").count(), 1, "the shared Scan must be merged onto one node, not duplicated" ); - assert!(graph.edge_annotations.is_empty()); + assert!(dag.edge_annotations.is_empty()); } /// Regression test: a single parent referencing the same shared child @@ -1511,26 +1511,26 @@ mod tests { left: Rc::clone(&shared_scan), right: Rc::clone(&shared_scan), }; - let graph = export_post_asap(&root, &mut |_| None); + let dag = export_post_asap(&root, &mut |_| None); assert_eq!( - graph.nodes.iter().filter(|n| n.kind == "Scan").count(), + dag.nodes.iter().filter(|n| n.kind == "Scan").count(), 1, "the shared Scan must be merged onto one node, not duplicated" ); assert!( - graph.edge_annotations.is_empty(), + dag.edge_annotations.is_empty(), "a single parent referencing the same child twice is one consumer, not a genuine \ multi-consumer share — got: {:?}", - graph.edge_annotations + dag.edge_annotations ); } #[test] fn export_never_produces_edge_annotations_since_it_never_shares_nodes() { // Plain `export` (no `export_post_asap`) never deduplicates by `Rc` - // pointer identity — even a workload-level shared subtree renders as - // two independent tree nodes here, so there is nothing to annotate. + // pointer identity — even a workload-level shared sub-DAG renders as + // two independent DAG nodes here, so there is nothing to annotate. let shared_scan = Rc::new(scan("metrics", value_col())); let root = QueryExpr::Join { kind: crate::pre_asap::query_expr::JoinKind::Inner, @@ -1538,9 +1538,9 @@ mod tests { left: Rc::clone(&shared_scan), right: Rc::clone(&shared_scan), }; - let graph = export(&root); - assert_eq!(graph.nodes.iter().filter(|n| n.kind == "Scan").count(), 2); - assert!(graph.edge_annotations.is_empty()); + let dag = export(&root); + assert_eq!(dag.nodes.iter().filter(|n| n.kind == "Scan").count(), 2); + assert!(dag.edge_annotations.is_empty()); } #[test] @@ -1558,18 +1558,18 @@ mod tests { child: Rc::new(scan("metrics", value_col())), }), }; - let graph = export(&expr); - assert_eq!(graph.nodes.len(), 3, "Filter -> Aggregate -> Scan"); + let dag = export(&expr); + assert_eq!(dag.nodes.len(), 3, "Filter -> Aggregate -> Scan"); - let filter = &graph.nodes[graph.root as usize]; + let filter = &dag.nodes[dag.root as usize]; assert_eq!(filter.kind, "Filter"); assert_eq!(filter.children.len(), 1); - let agg = &graph.nodes[filter.children[0] as usize]; + let agg = &dag.nodes[filter.children[0] as usize]; assert_eq!(agg.kind, "Aggregate"); assert_eq!(agg.children.len(), 1); - let leaf = &graph.nodes[agg.children[0] as usize]; + let leaf = &dag.nodes[agg.children[0] as usize]; assert_eq!(leaf.kind, "Scan"); assert!(leaf.children.is_empty()); } @@ -1581,37 +1581,37 @@ mod tests { scan("b", value_col()), scan("c", value_col()), ]); - let graph = export(&expr); - assert_eq!(graph.nodes.len(), 4, "3 branches + the Concat node"); - let merge = &graph.nodes[graph.root as usize]; + let dag = export(&expr); + assert_eq!(dag.nodes.len(), 4, "3 branches + the Concat node"); + let merge = &dag.nodes[dag.root as usize]; assert_eq!(merge.kind, "Concat"); assert_eq!(merge.children.len(), 3); } #[test] - fn identical_subtrees_hash_equal_and_differing_ones_dont() { + fn identical_sub_dags_hash_equal_and_differing_ones_dont() { let left = scan("metrics", value_col()); let right = scan("metrics", value_col()); let different = scan("other_table", value_col()); - let left_graph = export(&left); - let right_graph = export(&right); - let different_graph = export(&different); + let left_dag = export(&left); + let right_dag = export(&right); + let different_dag = export(&different); assert_eq!( - left_graph.nodes[left_graph.root as usize].hash, - right_graph.nodes[right_graph.root as usize].hash, + left_dag.nodes[left_dag.root as usize].hash, + right_dag.nodes[right_dag.root as usize].hash, "structurally identical Scans must hash equal" ); assert_ne!( - left_graph.nodes[left_graph.root as usize].hash, - different_graph.nodes[different_graph.root as usize].hash, + left_dag.nodes[left_dag.root as usize].hash, + different_dag.nodes[different_dag.root as usize].hash, "a different table_ref must not collide" ); } #[test] - fn shared_subtree_hash_matches_across_a_larger_tree() { + fn shared_sub_dag_hash_matches_across_a_larger_dag() { // Two roots that each wrap the *same* Scan shape in a different outer // node — the exported hash should still flag the shared Scan even // though it's embedded at different depths / under different parents. @@ -1651,19 +1651,19 @@ mod tests { // this exact node, because it's the same function call, not a // parallel reimplementation that happens to agree. let leaf = scan("metrics", value_col()); - let graph = export(&leaf); + let dag = export(&leaf); assert_eq!( - graph.nodes[graph.root as usize].hash, + dag.nodes[dag.root as usize].hash, Some(structural_hash(&leaf, &mut HashCache::new())), "dag_export's root hash must equal cse::structural_hash(&leaf, &mut HashCache::new()) directly" ); } #[test] - fn every_node_hash_matches_cse_structural_hash_on_its_own_subtree() { - // A multi-level tree: check the parity holds at every depth, not - // just the root — each `DagNode::hash` must equal - // `structural_hash` applied to the actual `QueryExpr` subtree that + fn every_node_hash_matches_cse_structural_hash_on_its_own_sub_dag() { + // A multi-level DAG: check the parity holds at every depth, not + // just the root — each `DAGNode::hash` must equal + // `structural_hash` applied to the actual `QueryExpr` sub-DAG that // node represents. let agg = QueryExpr::Aggregate { reduction: Reduction::Reduce(GroupKeys::none()), @@ -1680,20 +1680,20 @@ mod tests { child: Rc::new(agg.clone()), }; - let graph = export(&root); + let dag = export(&root); assert_eq!( - graph.nodes[graph.root as usize].hash, + dag.nodes[dag.root as usize].hash, Some(structural_hash(&root, &mut HashCache::new())), "Filter root hash must match cse::structural_hash(&root, &mut HashCache::new())" ); - let filter = &graph.nodes[graph.root as usize]; - let agg_node = &graph.nodes[filter.children[0] as usize]; + let filter = &dag.nodes[dag.root as usize]; + let agg_node = &dag.nodes[filter.children[0] as usize]; assert_eq!( agg_node.hash, Some(structural_hash(&agg, &mut HashCache::new())), "the exported Aggregate node's hash must match cse::structural_hash \ - on the Aggregate subtree it represents, not just the root" + on the Aggregate sub-DAG it represents, not just the root" ); } @@ -1772,9 +1772,9 @@ mod tests { }, guarantee: Some(guarantee), }; - let graph = export_summary(&root); - let json = serde_json::to_value(&graph).unwrap(); - let root_json = &json["nodes"][graph.root as usize]; + let dag = export_summary(&root); + let json = serde_json::to_value(&dag).unwrap(); + let root_json = &json["nodes"][dag.root as usize]; assert_eq!(root_json["guarantee"]["metric"], "rank"); assert_eq!(root_json["guarantee"]["bound"]["op"], "sum"); assert_eq!( @@ -1792,12 +1792,12 @@ mod tests { assert!(state.get("guarantee").is_none()); assert_eq!(json["nodes"][0]["guarantee"]["bound"]["op"], "zero"); - let named = NamedGraph { + let named = NamedDAG { name: "q".into(), source: None, - graph: export(&leaf), + dag: export(&leaf), replacements: vec![], - post_graph: None, + post_dag: None, workload_cost: None, rejections: vec![TargetRejection { target_pre_id: 0, @@ -1817,8 +1817,8 @@ mod tests { "unsupported_composition" ); assert_eq!(json["rejections"][0]["error"]["input_metrics"][0], "rank"); - // Additive: a graph with no rejections omits the key entirely. - let plain = NamedGraph { + // Additive: a DAG with no rejections omits the key entirely. + let plain = NamedDAG { rejections: vec![], ..named }; diff --git a/crates/types/src/post_asap/cse.rs b/crates/types/src/post_asap/cse.rs index 1872243f8..745758674 100644 --- a/crates/types/src/post_asap/cse.rs +++ b/crates/types/src/post_asap/cse.rs @@ -171,7 +171,7 @@ fn same_node(left: &SummaryNode, right: &SummaryNode) -> bool { expression_equal && left.schema == right.schema && same_value(&left.guarantee, &right.guarantee) } -/// Intern equal selected subtrees across roots while preserving every root ID. +/// Intern equal selected sub-DAGs across roots while preserving every root ID. /// /// Only structural equality is used: no grouping, parameter, accuracy or source /// coercions are performed. All roots must belong to the same data snapshot or @@ -184,7 +184,7 @@ fn same_node(left: &SummaryNode, right: &SummaryNode) -> bool { /// distinct count, entropy and L2. Candidate generation sizes a variant for /// the strictest sibling consumer so differing accuracy targets can reach /// identical states here. -pub fn share_common_summary_subtrees( +pub fn share_common_summary_sub_dags( roots: Vec<(Id, Rc)>, ) -> Vec<(Id, Rc)> { fn visit( @@ -269,7 +269,7 @@ mod tests { // Equal separately constructed roots preserve both IDs but share identity. #[test] fn shares_equal_roots_and_preserves_ids() { - let roots = share_common_summary_subtrees(vec![("a", leaf(1.0)), ("b", leaf(1.0))]); + let roots = share_common_summary_sub_dags(vec![("a", leaf(1.0)), ("b", leaf(1.0))]); assert_eq!(roots[0].0, "a"); assert_eq!(roots[1].0, "b"); assert!(Rc::ptr_eq(&roots[0].1, &roots[1].1)); @@ -289,7 +289,7 @@ mod tests { }, guarantee: None, }); - let roots = share_common_summary_subtrees(vec![(0, leaf(1.0)), (1, merge)]); + let roots = share_common_summary_sub_dags(vec![(0, leaf(1.0)), (1, merge)]); let SummaryExpr::SummaryMerge { children, .. } = &roots[1].1.expr else { panic!() }; @@ -302,7 +302,7 @@ mod tests { fn distinct_guarantees_and_values_are_not_shared() { let mut unknown = leaf(1.0).as_ref().clone(); unknown.guarantee = None; - let roots = share_common_summary_subtrees(vec![ + let roots = share_common_summary_sub_dags(vec![ (0, leaf(1.0)), (1, Rc::new(unknown)), (2, leaf(2.0)), @@ -317,7 +317,7 @@ mod tests { fn signed_zero_is_not_coalesced() { for values in [[0.0, -0.0], [-0.0, 0.0]] { let roots = - share_common_summary_subtrees(vec![(0, leaf(values[0])), (1, leaf(values[1]))]); + share_common_summary_sub_dags(vec![(0, leaf(values[0])), (1, leaf(values[1]))]); assert!(!Rc::ptr_eq(&roots[0].1, &roots[1].1)); for ((_, root), expected) in roots.iter().zip(values) { let SummaryExpr::KeepPreAsap(expr) = &root.expr else { @@ -347,10 +347,10 @@ mod tests { (f64::INFINITY, f64::NEG_INFINITY), (f64::NAN, f64::NAN), ] { - let roots = share_common_summary_subtrees(vec![(0, wrapped(a)), (1, wrapped(b))]); + let roots = share_common_summary_sub_dags(vec![(0, wrapped(a)), (1, wrapped(b))]); assert!(!Rc::ptr_eq(&roots[0].1, &roots[1].1)); } - let roots = share_common_summary_subtrees(vec![ + let roots = share_common_summary_sub_dags(vec![ (0, wrapped(f64::INFINITY)), (1, wrapped(f64::INFINITY)), ]); @@ -399,7 +399,7 @@ mod tests { guarantee: None, }) } - let roots = share_common_summary_subtrees(vec![ + let roots = share_common_summary_sub_dags(vec![ ("p95", readout(0.95, 0.01)), ("p99", readout(0.99, 0.01)), ("strict", readout(0.95, 0.001)), @@ -413,7 +413,7 @@ mod tests { assert!(!Rc::ptr_eq(&producer(&roots[0].1), &producer(&roots[2].1))); } - // Fifty unique input nodes must not require walking an expanded 2^24 tree. + // Fifty unique input nodes must not require walking an expanded 2^24 DAG. // The timeout is a coarse runaway guard, not a performance SLA. #[test] fn shared_diamond_does_not_expand_during_comparison() { @@ -445,7 +445,7 @@ mod tests { } current } - let roots = share_common_summary_subtrees(vec![(0, diamond()), (1, diamond())]); + let roots = share_common_summary_sub_dags(vec![(0, diamond()), (1, diamond())]); assert!(Rc::ptr_eq(&roots[0].1, &roots[1].1)); done.send(()).unwrap(); }); diff --git a/crates/types/src/post_asap/execution_data_state.rs b/crates/types/src/post_asap/execution_data_state.rs index 3c1c3e0c1..a9368377c 100644 --- a/crates/types/src/post_asap/execution_data_state.rs +++ b/crates/types/src/post_asap/execution_data_state.rs @@ -40,7 +40,7 @@ //! not do is stay ambiguous inside one mixed plan: the same `Rc` //! reached once as update input and once as query-time fallback is //! [`ExecutionDataStateError::AmbiguousKeepPreAsap`], because no single execution of that -//! subtree can serve both roles. +//! sub-DAG can serve both roles. use std::collections::HashMap; use std::rc::Rc; @@ -187,7 +187,7 @@ pub enum ExecutionDataStateError { /// One shared `KeepPreAsap` node reached both as update-path raw input /// and as a query-time fallback — see the module docs. #[error( - "KeepPreAsap subtree is data_state-ambiguous: reached as {first} and as {second} in the same \ + "KeepPreAsap sub-DAG is data_state-ambiguous: reached as {first} and as {second} in the same \ plan" )] AmbiguousKeepPreAsap { @@ -232,7 +232,7 @@ impl ExecutionDataStateAssignment { } /// Initial layout proposed by semantic realization, not a restriction on physical -/// operator placement. `PostAsapDag::with_execution_phases` assigns the final +/// operator placement. `PostAsapDAG::with_execution_phases` assigns the final /// phase independently of payload kind. Returns `None` for /// [`SummaryExpr::KeepPreAsap`], whose data_state is assigned by the edge reaching /// it (see the module docs). @@ -635,7 +635,7 @@ fn child_domain( Ok(avail) } None => { - // A raw pre-ASAP subtree executes at whichever data_state its consumer + // A raw pre-ASAP sub-DAG executes at whichever data_state its consumer // needs: update-path input for maintenance-time operation edges, // query-time fallback for a read-time edge. State-only edges // can't consume plain rows at all. @@ -1254,7 +1254,7 @@ mod tests { #[test] fn a_shared_keep_pre_asap_reached_in_two_domains_is_ambiguous() { - // One raw subtree used both as update input (under a SummaryAgg) and + // One raw sub-DAG used both as update input (under a SummaryAgg) and // as a query-time fallback (under an ExactRead) — no single // execution can serve both, so the plan is rejected. let shared = keep(); diff --git a/crates/types/src/post_asap/expr.rs b/crates/types/src/post_asap/expr.rs index 6c48114a2..8373f1843 100644 --- a/crates/types/src/post_asap/expr.rs +++ b/crates/types/src/post_asap/expr.rs @@ -104,7 +104,7 @@ pub struct SummaryNode { /// The machine-readable accuracy guarantee of the *value* this node /// produces (issue #172) — `Some` on every finalized, caller-visible /// value: a `SummaryEstimate` readout, an `ExactAggregate`-family - /// `SummaryAgg` (its state *is* the value), or a `KeepPreAsap` subtree + /// `SummaryAgg` (its state *is* the value), or a `KeepPreAsap` sub-DAG /// (executed exactly). `None` on raw summary state — a sketch-family /// `SummaryAgg`, `SummaryMerge`, `SummarySubtract`, `SummaryDelete`, /// `SummaryJoin` — whose guarantee only exists once something reads it @@ -121,13 +121,13 @@ pub struct SummaryNode { /// rules selectively replace logical aggregates and joins in the pre-ASAP /// `QueryExpr` with summary-bound counterparts. Final selection can retain /// supported read-time value operations around independently planned children; -/// other unsupported subtrees pass through as `KeepPreAsap(Rc)`. +/// other unsupported sub-DAGs pass through as `KeepPreAsap(Rc)`. /// /// Traversing from the root node yields a DAG; shared sub-expressions appear /// as multiple `Rc` references to the same `SummaryNode`. #[derive(Debug, Clone, PartialEq)] pub enum SummaryExpr { - /// A pre-ASAP subtree kept as-is because it has no selected implementation + /// A pre-ASAP sub-DAG kept as-is because it has no selected implementation /// or supported residual decomposition. Output schema is the inner node's /// schema, lifted to `SummarySchema` with all fields as /// `SummaryFamilyType::Plain`. diff --git a/crates/types/src/post_asap/guarantee.rs b/crates/types/src/post_asap/guarantee.rs index 88e070ce9..cfc6f3f01 100644 --- a/crates/types/src/post_asap/guarantee.rs +++ b/crates/types/src/post_asap/guarantee.rs @@ -17,7 +17,7 @@ //! //! [`ResultGuarantee`] is attached to a finalized, caller-visible value — //! [`super::SummaryNode::guarantee`] on a `SummaryEstimate` readout, an -//! exact accumulator, or a kept pre-ASAP subtree — never to raw summary +//! exact accumulator, or a kept pre-ASAP sub-DAG — never to raw summary //! state (a `SummaryAgg` sketch node carries `None`; its readout carries the //! guarantee). Its statement is: //! @@ -33,7 +33,7 @@ //! //! ## Why expressions, not numbers //! -//! [`BoundExpr`]/[`ProbabilityExpr`] are tiny serializable expression trees +//! [`BoundExpr`]/[`ProbabilityExpr`] are tiny serializable expression DAGs //! rather than bare `f64`s so a planning-time guarantee can reference a //! statistic it does not have (a group count, a stream's L1 norm) and stay //! honestly *unknown* until something instantiates it — a deployment's own diff --git a/crates/types/src/post_asap/mod.rs b/crates/types/src/post_asap/mod.rs index 6bf317420..4c35559b4 100644 --- a/crates/types/src/post_asap/mod.rs +++ b/crates/types/src/post_asap/mod.rs @@ -40,7 +40,7 @@ pub mod summary_maintenance; pub mod summary_maintenance_lifecycle; pub mod summary_window; -pub use cse::share_common_summary_subtrees; +pub use cse::share_common_summary_sub_dags; pub use execution_data_state::{ assigned_child_data_state, exact_operation_output_schema, produced_data_state, validate_execution_data_states, validate_execution_data_states_at, DataPrimitive, @@ -56,8 +56,8 @@ pub use guarantee::{ }; pub use post_asap_dag::{ compile_post_asap_dag, compile_post_asap_dag_with_node_ids, EdgeRole, - GroupingEdgeCompatibility, PostAsapDag, PostAsapDagCompilation, PostAsapDagDocument, - PostAsapDagEdge, PostAsapDagNode, PostAsapDagValidationError, PostAsapNodeId, + GroupingEdgeCompatibility, PostAsapDAG, PostAsapDAGCompilation, PostAsapDAGDocument, + PostAsapDAGEdge, PostAsapDAGNode, PostAsapDAGValidationError, PostAsapNodeId, PostAsapNodeIdentityMap, PostAsapOperatorPayload, WindowEdgeCompatibility, POST_ASAP_DAG_WIRE_VERSION, }; diff --git a/crates/types/src/post_asap/post_asap_dag.rs b/crates/types/src/post_asap/post_asap_dag.rs index adaabb1b3..ebbe51554 100644 --- a/crates/types/src/post_asap/post_asap_dag.rs +++ b/crates/types/src/post_asap/post_asap_dag.rs @@ -90,7 +90,7 @@ pub enum PostAsapOperatorPayload { #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] -pub struct PostAsapDagNode { +pub struct PostAsapDAGNode { pub id: PostAsapNodeId, /// The payload variant is the sole operator identity (`payload.kind` in JSON). pub payload: PostAsapOperatorPayload, @@ -102,7 +102,7 @@ pub struct PostAsapDagNode { #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] -pub struct PostAsapDagEdge { +pub struct PostAsapDAGEdge { pub producer: PostAsapNodeId, pub consumer: PostAsapNodeId, pub role: EdgeRole, @@ -114,9 +114,9 @@ pub struct PostAsapDagEdge { #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] -pub struct PostAsapDag { - pub nodes: Vec, - pub edges: Vec, +pub struct PostAsapDAG { + pub nodes: Vec, + pub edges: Vec, /// Semantic workload root. Physical query/precompute sinks are selected /// downstream by the control plane. pub root: PostAsapNodeId, @@ -127,13 +127,13 @@ pub struct PostAsapDag { /// Process boundaries exchange this envelope and call [`Self::validate`]. #[derive(Debug, Clone, PartialEq, serde::Serialize, serde::Deserialize)] #[serde(deny_unknown_fields)] -pub struct PostAsapDagDocument { +pub struct PostAsapDAGDocument { pub schema_version: u32, - pub dag: PostAsapDag, + pub dag: PostAsapDAG, } #[derive(Debug, Clone, PartialEq, Eq, Error)] -pub enum PostAsapDagValidationError { +pub enum PostAsapDAGValidationError { #[error("phase assignment must name every DAG node exactly once")] IncompletePhaseAssignment, #[error("ingestion node {consumer:?} depends on query node {producer:?}")] @@ -171,17 +171,17 @@ pub enum PostAsapDagValidationError { SummaryGroupingMismatch { node: PostAsapNodeId }, } -impl PostAsapDagDocument { - pub fn new(dag: PostAsapDag) -> Self { +impl PostAsapDAGDocument { + pub fn new(dag: PostAsapDAG) -> Self { Self { schema_version: POST_ASAP_DAG_WIRE_VERSION, dag, } } - pub fn validate(&self) -> Result<(), PostAsapDagValidationError> { + pub fn validate(&self) -> Result<(), PostAsapDAGValidationError> { if self.schema_version != POST_ASAP_DAG_WIRE_VERSION { - return Err(PostAsapDagValidationError::UnsupportedVersion( + return Err(PostAsapDAGValidationError::UnsupportedVersion( self.schema_version, )); } @@ -189,19 +189,19 @@ impl PostAsapDagDocument { } } -impl PostAsapDag { +impl PostAsapDAG { /// Assign execution phases without changing operator semantics. Phase choices /// do not prove deployment support: callers must bind concrete implementations /// and storage boundaries before installing this plan. pub fn with_execution_phases( &self, phases: &std::collections::BTreeMap, - ) -> Result { + ) -> Result { self.validate()?; if phases.len() != self.nodes.len() || self.nodes.iter().any(|node| !phases.contains_key(&node.id)) { - return Err(PostAsapDagValidationError::IncompletePhaseAssignment); + return Err(PostAsapDAGValidationError::IncompletePhaseAssignment); } let mut dag = self.clone(); for node in &mut dag.nodes { @@ -215,12 +215,12 @@ impl PostAsapDag { Ok(dag) } - pub fn validate(&self) -> Result<(), PostAsapDagValidationError> { + pub fn validate(&self) -> Result<(), PostAsapDAGValidationError> { use std::collections::{HashMap, HashSet}; let mut nodes = HashMap::new(); for node in &self.nodes { if nodes.insert(node.id, node).is_some() { - return Err(PostAsapDagValidationError::DuplicateNodeId(node.id)); + return Err(PostAsapDAGValidationError::DuplicateNodeId(node.id)); } if let PostAsapOperatorPayload::SummaryAgg { family, grouping, .. @@ -233,48 +233,48 @@ impl PostAsapDag { } if let SummaryFamilyType::Sketch(_, schema_grouping) = &field.dtype { if schema_grouping != grouping { - return Err(PostAsapDagValidationError::SummaryGroupingMismatch { + return Err(PostAsapDAGValidationError::SummaryGroupingMismatch { node: node.id, }); } } } if !found_family { - return Err(PostAsapDagValidationError::SummaryFamilySchemaMismatch { + return Err(PostAsapDAGValidationError::SummaryFamilySchemaMismatch { node: node.id, }); } } } if !nodes.contains_key(&self.root) { - return Err(PostAsapDagValidationError::MissingRoot(self.root)); + return Err(PostAsapDAGValidationError::MissingRoot(self.root)); } let mut children: HashMap> = HashMap::new(); for edge in &self.edges { let producer = nodes.get(&edge.producer).ok_or( - PostAsapDagValidationError::MissingEdgeEndpoint(edge.producer), + PostAsapDAGValidationError::MissingEdgeEndpoint(edge.producer), )?; if !nodes.contains_key(&edge.consumer) { - return Err(PostAsapDagValidationError::MissingEdgeEndpoint( + return Err(PostAsapDAGValidationError::MissingEdgeEndpoint( edge.consumer, )); } if producer.output_state.timing == ExecutionTiming::QueryTime && nodes[&edge.consumer].output_state.timing == ExecutionTiming::IngestionTime { - return Err(PostAsapDagValidationError::QueryDependencyInIngestion { + return Err(PostAsapDAGValidationError::QueryDependencyInIngestion { producer: edge.producer, consumer: edge.consumer, }); } if edge.intermediate_schema != producer.output_schema { - return Err(PostAsapDagValidationError::EdgeSchemaMismatch { + return Err(PostAsapDAGValidationError::EdgeSchemaMismatch { producer: edge.producer, consumer: edge.consumer, }); } if edge.data_state != producer.output_state { - return Err(PostAsapDagValidationError::EdgeDataStateMismatch { + return Err(PostAsapDAGValidationError::EdgeDataStateMismatch { producer: edge.producer, consumer: edge.consumer, }); @@ -314,7 +314,7 @@ impl PostAsapDag { &mut HashSet::new(), &mut HashSet::new(), ) { - return Err(PostAsapDagValidationError::Cycle); + return Err(PostAsapDAGValidationError::Cycle); } let mut reachable = HashSet::new(); fn mark( @@ -331,7 +331,7 @@ impl PostAsapDag { } mark(self.root, &children, &mut reachable); if let Some(id) = nodes.keys().find(|id| !reachable.contains(id)) { - return Err(PostAsapDagValidationError::UnreachableNode(*id)); + return Err(PostAsapDAGValidationError::UnreachableNode(*id)); } Ok(()) } @@ -359,20 +359,20 @@ impl PostAsapNodeIdentityMap { } #[derive(Debug, Clone)] -pub struct PostAsapDagCompilation { - pub dag: PostAsapDag, +pub struct PostAsapDAGCompilation { + pub dag: PostAsapDAG, pub node_ids: PostAsapNodeIdentityMap, } pub fn compile_post_asap_dag( root: &Rc, -) -> Result { +) -> Result { Ok(compile_post_asap_dag_with_node_ids(root)?.dag) } pub fn compile_post_asap_dag_with_node_ids( root: &Rc, -) -> Result { +) -> Result { let assignment = validate_execution_data_states(root)?; let mut nodes = Vec::new(); let mut edges = Vec::new(); @@ -383,8 +383,8 @@ pub fn compile_post_asap_dag_with_node_ids( node: &Rc, assignment: &super::ExecutionDataStateAssignment, ids: &mut HashMap<*const SummaryNode, PostAsapNodeId>, - nodes: &mut Vec, - edges: &mut Vec, + nodes: &mut Vec, + edges: &mut Vec, nodes_by_id: &mut Vec>, ) -> PostAsapNodeId { if let Some(id) = ids.get(&Rc::as_ptr(node)) { @@ -474,7 +474,7 @@ pub fn compile_post_asap_dag_with_node_ids( } SummaryExpr::SummaryMerge { .. } => PostAsapOperatorPayload::SummaryMerge, }; - nodes.push(PostAsapDagNode { + nodes.push(PostAsapDAGNode { id, payload, output_state: state, @@ -528,12 +528,12 @@ pub fn compile_post_asap_dag_with_node_ids( } _ => GroupingEdgeCompatibility::NotApplicable, }; - edges.push(PostAsapDagEdge { + edges.push(PostAsapDAGEdge { producer, consumer: id, role, intermediate_schema: child.schema.clone(), - // The whole-graph validator owns contextual state assignment, + // The whole-DAG validator owns contextual state assignment, // especially for shared KeepPreAsap leaves. Export that // authoritative result instead of independently deriving the // edge state a second time. @@ -559,10 +559,10 @@ pub fn compile_post_asap_dag_with_node_ids( &mut edges, &mut nodes_by_id, ); - let dag = PostAsapDag { nodes, edges, root }; + let dag = PostAsapDAG { nodes, edges, root }; dag.validate() .expect("compiler emits a valid post-ASAP DAG"); - Ok(PostAsapDagCompilation { + Ok(PostAsapDAGCompilation { dag, node_ids: PostAsapNodeIdentityMap { nodes_by_id }, }) @@ -643,10 +643,10 @@ mod tests { | PostAsapOperatorPayload::SummaryDelete { .. } | PostAsapOperatorPayload::SummaryMerge => DataPrimitive::SummaryState, }; - let dag = PostAsapDag { + let dag = PostAsapDAG { root: PostAsapNodeId(0), edges: vec![], - nodes: vec![PostAsapDagNode { + nodes: vec![PostAsapDAGNode { id: PostAsapNodeId(0), payload: payload.clone(), output_state: ExecutionDataState { @@ -672,7 +672,7 @@ mod tests { assert_eq!(placed.nodes[0].output_state.timing, phase); let wire = serde_json::to_value(&placed).unwrap(); assert!(wire["nodes"][0]["payload"].get("timing").is_none()); - assert_eq!(serde_json::from_value::(wire).unwrap(), placed); + assert_eq!(serde_json::from_value::(wire).unwrap(), placed); } assert!(dag.with_execution_phases(&BTreeMap::new()).is_err()); } @@ -687,7 +687,7 @@ mod tests { }; let nodes = [0, 1] .into_iter() - .map(|id| PostAsapDagNode { + .map(|id| PostAsapDAGNode { id: PostAsapNodeId(id), payload: PostAsapOperatorPayload::Fallback { expression: QueryExpr::Literal(ScalarValue::Int64(1)), @@ -697,10 +697,10 @@ mod tests { guarantee: None, }) .collect(); - let dag = PostAsapDag { + let dag = PostAsapDAG { nodes, root: PostAsapNodeId(1), - edges: vec![PostAsapDagEdge { + edges: vec![PostAsapDAGEdge { producer: PostAsapNodeId(0), consumer: PostAsapNodeId(1), role: EdgeRole::Input, @@ -726,7 +726,7 @@ mod tests { (PostAsapNodeId(0), ExecutionTiming::QueryTime), (PostAsapNodeId(1), ExecutionTiming::IngestionTime), ])), - Err(PostAsapDagValidationError::QueryDependencyInIngestion { .. }) + Err(PostAsapDAGValidationError::QueryDependencyInIngestion { .. }) )); } @@ -822,14 +822,14 @@ mod tests { SummaryFamilyType::ExactAggregate(ExactKind::Sum, _) )); let encoded = serde_json::to_string(&dag).expect("serialize post-ASAP DAG"); - let decoded: PostAsapDag = + let decoded: PostAsapDAG = serde_json::from_str(&encoded).expect("deserialize post-ASAP DAG"); assert_eq!(decoded, dag); - let document = PostAsapDagDocument::new(decoded); + let document = PostAsapDAGDocument::new(decoded); document.validate().unwrap(); let mut invalid = serde_json::to_value(&document).unwrap(); invalid["dag"]["nodes"][0]["operator"] = serde_json::json!("Binary"); - assert!(serde_json::from_value::(invalid).is_err()); + assert!(serde_json::from_value::(invalid).is_err()); assert!(document.dag.nodes.iter().all(|node| { let wire = serde_json::to_value(node).unwrap(); wire.get("operator").is_none() && wire["payload"]["kind"].is_string() @@ -838,11 +838,11 @@ mod tests { old_version.schema_version = 1; assert_eq!( old_version.validate(), - Err(PostAsapDagValidationError::UnsupportedVersion(1)) + Err(PostAsapDAGValidationError::UnsupportedVersion(1)) ); let mut unknown = serde_json::to_value(&document).unwrap(); unknown["unexpected"] = serde_json::json!(true); - assert!(serde_json::from_value::(unknown).is_err()); + assert!(serde_json::from_value::(unknown).is_err()); assert!(matches!( dag.nodes[2].payload, PostAsapOperatorPayload::SummaryAgg { diff --git a/crates/types/src/post_asap/sketch.rs b/crates/types/src/post_asap/sketch.rs index 391d12f0f..416341e73 100644 --- a/crates/types/src/post_asap/sketch.rs +++ b/crates/types/src/post_asap/sketch.rs @@ -606,7 +606,7 @@ pub enum SketchQuery { /// `count(cms_metric{item="checkout"})` — `key` is `item`, `value` is /// `"checkout"`). `value` is carried here rather than resolved by the /// `SummaryExecutor` from a `Filter` predicate because `readout`'s - /// trait signature has no tree access — see `CostModel::readout_extension`. + /// trait signature has no DAG access — see `CostModel::readout_extension`. PointCount { key: ColumnRef, value: Option, diff --git a/crates/types/src/pre_asap/agg_intent.rs b/crates/types/src/pre_asap/agg_intent.rs index 2203f4ec7..5e4079b2e 100644 --- a/crates/types/src/pre_asap/agg_intent.rs +++ b/crates/types/src/pre_asap/agg_intent.rs @@ -380,7 +380,7 @@ pub enum MathFunc { // `requires` / `is_per_series` / `output_column` never read `col`'s value — // only its presence via a `{ .. }` pattern — so, unlike // `QueryExpr::output_schema` (which genuinely cannot compile for an -// unresolved tree — see its own doc), nothing stops these from being generic +// unresolved DAG — see its own doc), nothing stops these from being generic // over every `C`. And a front end constructing `AggIntent` // directly (issue #179) does need `is_per_series` pre-binding — it decides // the `PerEntity`/`Reduce` reduction shape right at construction time (see @@ -930,7 +930,7 @@ mod tests { } } - /// `col` is `#[serde(default)]`, so a tree serialized before issue #115 — + /// `col` is `#[serde(default)]`, so a DAG serialized before issue #115 — /// with no `col` key — still deserializes, as the sample-value convention `None`. #[test] fn agg_intent_serde_reads_pre_115_payloads() { diff --git a/crates/types/src/pre_asap/canonicalize.rs b/crates/types/src/pre_asap/canonicalize.rs index 3a66c5c4d..4f15dcebb 100644 --- a/crates/types/src/pre_asap/canonicalize.rs +++ b/crates/types/src/pre_asap/canonicalize.rs @@ -1,7 +1,7 @@ //! Shared post-lowering canonicalization of the resolved [`QueryExpr`]. //! //! Both language front ends funnel through [`resolve_root`](super::resolve::resolve_root), -//! which runs this pass over the resolved tree. Its job is to erase +//! which runs this pass over the resolved DAG. Its job is to erase //! *structural* differences between semantically identical queries so a //! post-ASAP binding rule matching on the intent algebra sees one canonical //! spelling regardless of source language (issue #34). @@ -29,7 +29,7 @@ use super::expr_ir::{CompareOpKind, ScalarValue}; use super::query_expr::{Predicate, QueryExpr, Reduction, SortKey, WindowFuncKind}; use crate::types::AccuracyTarget; -/// Rewrite `expr` into its canonical form (bottom-up). Idempotent: a tree that +/// Rewrite `expr` into its canonical form (bottom-up). Idempotent: a DAG that /// is already canonical is returned unchanged. pub fn canonicalize(mut expr: QueryExpr) -> QueryExpr { canon(&mut expr); @@ -48,7 +48,7 @@ fn canon(expr: &mut QueryExpr) { // pointing at the wrong column, or out of bounds, of the // post-canonicalize schema. Snapshot the schema the discriminator key // was actually resolved against, right here, before recursing into the - // children — this is the exact tree state `resolve.rs` saw. + // children — this is the exact DAG state `resolve.rs` saw. let discriminator_branch_schema_before = match expr { QueryExpr::Concat { children, @@ -96,7 +96,7 @@ fn canon(expr: &mut QueryExpr) { /// A `&mut QueryExpr` out of a child `Rc` — clone-on-write via /// [`Rc::make_mut`]: free (no clone) while `r` is uniquely owned, which is -/// the overwhelmingly common case (a tree `canonicalize` was just handed by +/// the overwhelmingly common case (a DAG `canonicalize` was just handed by /// value); falls back to cloning just *this* node (its own fields — the /// grandchildren stay shared `Rc`s, not deep-copied) only when some other /// owner still holds the same `Rc`, e.g. a caller that kept its own clone @@ -106,7 +106,7 @@ fn canon(expr: &mut QueryExpr) { /// panic on exactly that case; `make_mut` degrades to a shallow copy instead /// of requiring sole ownership as a precondition. Once a workload-level CSE /// pass runs (issue #212, #222) and canonicalize sees an already-shared -/// subtree from a *different* query, this is also the mechanism that keeps +/// sub-DAG from a *different* query, this is also the mechanism that keeps /// canonicalizing one query from silently corrupting another's view of the /// same shared node. fn rc_mut(r: &mut Rc) -> &mut QueryExpr { @@ -117,7 +117,7 @@ fn rc_mut(r: &mut Rc) -> &mut QueryExpr { /// node — `canon`'s own top-down/bottom-up walk only ever visits the /// relational skeleton, never descending into a scalar position (`Filter.pred`, /// `ProjectItem.expr`, …): none of the three rewrite rules rewrite anything -/// inside a scalar subtree, so there's nothing to gain by recursing into one, +/// inside a scalar sub-DAG, so there's nothing to gain by recursing into one, /// and every scalar variant (issue #205) hits the catch-all below. fn children_mut(expr: &mut QueryExpr) -> Vec<&mut QueryExpr> { use QueryExpr::*; diff --git a/crates/types/src/pre_asap/column_resolution.rs b/crates/types/src/pre_asap/column_resolution.rs index 6f974922e..3176730a9 100644 --- a/crates/types/src/pre_asap/column_resolution.rs +++ b/crates/types/src/pre_asap/column_resolution.rs @@ -1,7 +1,7 @@ //! Schema-driven column resolution. //! //! Front ends (issue #179) emit `ColumnRef` (name-based, optionally -//! table-qualified); the canonical tree uses positional [`ColumnId`] resolved +//! table-qualified); the canonical DAG uses positional [`ColumnId`] resolved //! against a per-node [`Schema`]. These helpers bridge the two — the //! [`SchemaResolver`](super::schema_resolver) builds the schema, and [`resolve_column_refs`] //! turns name-based refs (group keys, dedup columns) into positional ids, diff --git a/crates/types/src/pre_asap/cse.rs b/crates/types/src/pre_asap/cse.rs index 1a2758737..45842bf1d 100644 --- a/crates/types/src/pre_asap/cse.rs +++ b/crates/types/src/pre_asap/cse.rs @@ -1,14 +1,14 @@ //! Pre-ASAP structural common-subexpression elimination: bottom-up -//! hash-consing over an already-`resolve_root`'d [`QueryExpr`] tree (issue +//! hash-consing over an already-`resolve_root`'d [`QueryExpr`] DAG (issue //! #212, #222, #223). //! -//! CSE only runs on an already-bound, already-canonicalized tree — +//! CSE only runs on an already-bound, already-canonicalized DAG — //! structural matching is meaningless before canonicalization has converged //! semantically-equivalent queries onto one shape (`docs/develop_docs/pre-asap-ir.md` //! design principle 3; `median(latency)` and `approx_percentile_cont(latency, //! 0.5)` already lower to an identical `AggIntent::Quantile` today, per //! `sql_lowering.rs`'s `median_is_the_same_intent_as_an_explicit_half_percentile` -//! test). [`share_common_subtrees`] is the single entry point, run once per +//! test). [`share_common_sub_dags`] is the single entry point, run once per //! workload batch (or once per query — see "Single-query CSE" below) *after* //! `resolve_root`, *before* the pre-ASAP → post-ASAP replacement/search pass //! (`asap_aware_mapping::replacement`). @@ -19,7 +19,7 @@ //! candidacy for sharing naturally incorporates whether its own children were //! themselves shared — two parents whose children were independently //! deduplicated down to the same `Rc`s are structurally identical iff their -//! own fields also match, without re-walking the subtrees. +//! own fields also match, without re-walking the sub-DAGs. //! //! Only the **relational skeleton** participates — the same set of "operator" //! children [`canonicalize`](super::canonicalize)'s `children_mut` walks @@ -30,13 +30,13 @@ //! `QueryExpr`'s derived `PartialEq` along with the rest of that node's //! fields, rather than separately hash-consed — the same scope //! `canonicalize.rs` settled on ("none of the rewrite rules touch a scalar -//! subtree, so there's nothing to gain by recursing into one"). Widening this +//! sub-DAG, so there's nothing to gain by recursing into one"). Widening this //! to scalar positions is future work, not attempted here. //! //! ## Correctness: hash is a filter, `PartialEq` is the decision //! //! This is the one non-negotiable rule. A **false positive** here — two -//! subtrees wrongly judged shareable — is a wrong query answer, not a missed +//! sub-DAGs wrongly judged shareable — is a wrong query answer, not a missed //! optimization: two different queries would read each other's data. //! [`structural_hash`] (`DefaultHasher`/SipHash over a canonical //! serialization, no collision-freedom guarantee) may only narrow the @@ -75,23 +75,23 @@ //! A repeated sub-expression within *one* query (e.g. the same grouped //! `Aggregate` referenced twice on two `BinaryOp` branches) is deduplicated //! by the exact same bottom-up interning — a workload of size one still -//! interns bottom-up within that one tree. No separate mechanism is needed; -//! see the `single_query_shares_its_own_repeated_subtree` test below. +//! interns bottom-up within that one DAG. No separate mechanism is needed; +//! see the `single_query_shares_its_own_repeated_sub_dag` test below. //! //! ## Landing plan (issue #223) //! //! This module is stage 1 of a 4-stage plan. Stage 2 //! (`asap_aware_mapping::replacement::search_workload_with`, which runs -//! [`share_common_subtrees`] itself before searching) is a real caller, +//! [`share_common_sub_dags`] itself before searching) is a real caller, //! wired at the same time so this never becomes unwired dead code again //! (the original `asap-plan::cse::dedupe_subtrees` was deleted in #192 for //! exactly that). Stage 3 — [`dag_export`](crate::dag_export) computing its //! per-node `hash` by calling this module's [`structural_hash`] directly, //! instead of a parallel reimplementation — is also done, so -//! `tools/dag-viewer`'s "shared subtree" highlighting now flags exactly the +//! `tools/dag-viewer`'s "shared sub-DAG" highlighting now flags exactly the //! candidate pairs this module's own `InternTable` would bucket together //! (still only a hash match, not a guarantee of -//! `share_common_subtrees`-actual sharing — see `dag_export`'s module doc). +//! `share_common_sub_dags`-actual sharing — see `dag_export`'s module doc). //! Stage 4 (issue #237) is implemented in //! `asap_aware_mapping::cost_model::CostModel::cse_share_decision`, called //! from `asap_aware_mapping::replacement::CandidateLogicalASAPDAGs::cost_sorted` (via that @@ -187,27 +187,27 @@ pub type HashCache = HashMap<*const QueryExpr, u64>; /// up in `cache` if already computed there (memoized by `Rc` pointer /// identity) rather than recursed into again. /// -/// This is the DAG-aware fix a naive "just serialize the whole subtree" -/// hash would get wrong: after [`share_common_subtrees`] (or even before +/// This is the DAG-aware fix a naive "just serialize the whole sub-DAG" +/// hash would get wrong: after [`share_common_sub_dags`] (or even before /// it — a front end can emit internal `Rc` sharing directly, e.g. a -/// repeated subexpression within one query), `node` is generally a DAG, -/// not a tree. A full-subtree serialization re-serializes — re-walks — +/// repeated subexpression within one query), `node` generally has internal +/// sharing. A full-sub-DAG serialization re-serializes — re-walks — /// any descendant `node` already shares internally once per parent that /// references it; called once per node in a bottom-up pass (as /// [`InternTable::intern`] and [`dag_export`](crate::dag_export) both do), -/// that costs `O(subtree size)` *per node* instead of `O(1)` amortized — +/// that costs `O(sub-DAG size)` *per node* instead of `O(1)` amortized — /// quadratic-or-worse for a deep chain, compounding further with any real /// internal sharing. Memoizing each child's hash by pointer identity in /// `cache` (persisted across the whole pass by the caller, not reset per /// node) makes each node's own contribution `O(1)` beyond its children's /// already-known hashes, giving `O(N)` total for `N` nodes — matching -/// [`dag_node_count`]'s own DAG-vs-tree fix (issue #212/#223/#237's stage +/// [`dag_node_count`]'s own shared-node counting fix (issue #212/#223/#237's stage /// 4) in spirit, applied to hashing instead of counting. /// /// `pub` (not private) so [`dag_export`](crate::dag_export) can call /// this exact function for its exported nodes' `hash` field instead of /// maintaining its own parallel reimplementation — issue #223 stage 3. That -/// makes `tools/dag-viewer`'s "shared subtree" highlighting reflect this +/// makes `tools/dag-viewer`'s "shared sub-DAG" highlighting reflect this /// module's real hashing, not a lookalike computed a different way; see the /// module doc's "Landing plan" section. A NaN/infinite `f64` makes JSON /// serialization fail; falling back to a fixed hash just puts every such @@ -236,9 +236,9 @@ pub fn structural_hash(node: &QueryExpr, cache: &mut HashCache) -> u64 { /// Hash `own_fields` (this node's own tag and non-child scalar /// fields — anything JSON-serializable and small, i.e. never a - /// `QueryExpr` subtree) via the same canonical-JSON-string trick the - /// whole-subtree version used, just applied to `O(1)` fields instead - /// of `O(subtree size)`. + /// `QueryExpr` sub-DAG) via the same canonical-JSON-string trick the + /// whole-sub-DAG version used, just applied to `O(1)` fields instead + /// of `O(sub-DAG size)`. fn hash_own_fields(hasher: &mut impl Hasher, own_fields: &impl serde::Serialize) { serde_json::to_string(own_fields) .unwrap_or_default() @@ -408,7 +408,7 @@ pub fn structural_hash(node: &QueryExpr, cache: &mut HashCache) -> u64 { // (issue #205) are all leaves for this traversal's purposes — none // has an operator child to look up in `cache` — so hashing the // whole node via `serde_json` in one shot is already `O(node - // size)`, not `O(subtree size)`: exactly the same cost the + // size)`, not `O(sub-DAG size)`: exactly the same cost the // per-variant `hash_own_fields` calls above pay, just without // needing to spell out each field individually. Matches // `rebuild_children`'s and `dag_node_count`'s identical scope @@ -435,13 +435,13 @@ pub fn structural_hash(node: &QueryExpr, cache: &mut HashCache) -> u64 { /// Count of *unique* nodes reachable from `root`, deduplicated by `Rc` /// pointer identity (`Rc::as_ptr`) — the real size of the DAG rooted at -/// `root`, not a tree-walk count. +/// `root`, not a per-path walk count. /// -/// After [`share_common_subtrees`] runs (or even before it, for a tree a +/// After [`share_common_sub_dags`] runs (or even before it, for a DAG a /// front end already built with internal `Rc` sharing — e.g. re-running /// CSE, or a single-query repeated subexpression), `root` is generally a -/// **DAG**, not a tree — that is this whole module's premise. Anything that -/// walks `root` as if every reference were a fresh subtree (a naive +/// **DAG** with internal sharing — that is this whole module's premise. Anything that +/// walks `root` as if every reference were a fresh sub-DAG (a naive /// recursive walk with no identity tracking, or a naive full /// `serde_json` serialization — `Rc`'s `Serialize` impl serializes the /// pointee's *value* at every occurrence, it does not dedupe by identity) @@ -454,9 +454,9 @@ pub fn structural_hash(node: &QueryExpr, cache: &mut HashCache) -> u64 { /// `pub` so cost-aware callers outside this crate (e.g. /// `asap_aware_mapping::CostModel::cse_recompute_cost`'s default) have a /// DAG-correct structural-size proxy available, instead of reaching for -/// something tree-shaped like a raw serialization length. +/// something per-path like a raw serialization length. /// -/// Same operator-child traversal scope as [`share_common_subtrees`] itself +/// Same operator-child traversal scope as [`share_common_sub_dags`] itself /// (see the module doc's "Algorithm" section, and this module's private /// `rebuild_children`) — a scalar subexpression embedded in a wrapper /// position (`Predicate`, `ProjectItem.expr`, `Aggregate.having`, …) is not @@ -542,11 +542,11 @@ fn count_unique(node: &QueryExpr, seen: &mut std::collections::HashSet<*const Qu /// Recurse into `child`, then intern the result. `Rc::try_unwrap` recovers /// the owned node without cloning in the overwhelmingly common case — a -/// tree freshly built by a front end / `resolve_root`, not yet shared by any +/// DAG freshly built by a front end / `resolve_root`, not yet shared by any /// prior CSE pass, where every `Rc` is uniquely owned. Falls back to cloning /// this node's own fields (its children stay `Rc`s, not deep-copied) only -/// when `child` is already shared — e.g. re-running CSE over a tree that -/// went through a previous `share_common_subtrees` pass; a structural +/// when `child` is already shared — e.g. re-running CSE over a DAG that +/// went through a previous `share_common_sub_dags` pass; a structural /// duplicate collapses right back onto `child` itself via `PartialEq`, an /// already-optimal no-op. fn intern_child(table: &mut InternTable, child: Rc) -> Rc { @@ -750,17 +750,17 @@ fn rebuild_children(table: &mut InternTable, expr: QueryExpr) -> QueryExpr { } } -/// Share structurally-identical, sharing-legal subtrees across a workload's +/// Share structurally-identical, sharing-legal sub-DAGs across a workload's /// query roots (or within one query, for `roots.len() == 1` — see the /// module doc's "Single-query CSE" section). Every root's *value* is /// unchanged (`PartialEq`-equal to its input) — only its internal `Rc` -/// structure may now alias another root's, or another part of its own tree. +/// structure may now alias another root's, or another part of its own DAG. /// /// `roots` must already be bound + canonicalized (post-`resolve_root`). /// `Id` is caller-chosen — a `QueryWorkload` entry's own key, an index, a /// query name, whatever identifies one root through the pipeline; this /// module has no opinion on its shape. -pub fn share_common_subtrees(roots: Vec<(Id, QueryExpr)>) -> Vec<(Id, Rc)> { +pub fn share_common_sub_dags(roots: Vec<(Id, QueryExpr)>) -> Vec<(Id, Rc)> { let mut table = InternTable::new(); roots .into_iter() @@ -816,7 +816,7 @@ mod tests { // blocking the merge — only the differing `col` is. let a = quantile_agg(vec![1], Some(2), 0.5); let b = quantile_agg(vec![1], Some(3), 0.5); - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); + let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; @@ -841,7 +841,7 @@ mod tests { op: CompareOpKind::Gt, right: Rc::new(QueryExpr::Literal(ScalarValue::Float64(1.0))), })))]; - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); + let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; @@ -856,12 +856,12 @@ mod tests { // hoistable even though `a` and `b` are structurally identical. let a = quantile_agg(vec![], Some(2), 0.9); let b = quantile_agg(vec![], Some(2), 0.9); - assert_eq!(a, b, "fixture sanity: the two trees are structurally equal"); + assert_eq!(a, b, "fixture sanity: the two DAGs are structurally equal"); assert!( !a.output_schema().unwrap().has_unique_key(), "fixture sanity: an ungrouped aggregate has no provable unique key" ); - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); + let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; @@ -878,11 +878,11 @@ mod tests { // 0.5, .. }` today (see `sql_lowering.rs`'s // `median_is_the_same_intent_as_an_explicit_half_percentile`) — here // built directly (grouped, so a unique key is provable) as two - // independently-constructed but structurally identical trees, the + // independently-constructed but structurally identical DAGs, the // way two different call sites in a workload would produce them. let median = quantile_agg(vec![1], Some(2), 0.5); let approx_percentile_cont_half = quantile_agg(vec![1], Some(2), 0.5); - let shared = share_common_subtrees(vec![ + let shared = share_common_sub_dags(vec![ ("median", median), ("percentile", approx_percentile_cont_half), ]); @@ -896,13 +896,13 @@ mod tests { } #[test] - fn single_query_shares_its_own_repeated_subtree() { + fn single_query_shares_its_own_repeated_sub_dag() { // One query root referencing the same grouped Aggregate on both // BinaryOp branches — built as two separately-allocated but - // structurally identical subtrees (`.clone()` into two distinct + // structurally identical sub-DAGs (`.clone()` into two distinct // `Rc::new` calls), the shape a front end emitting a repeated // sub-expression would actually produce (no sharing yet). A - // workload of size 1 still interns bottom-up within this one tree — + // workload of size 1 still interns bottom-up within this one DAG — // no separate single-query mechanism needed. let agg = quantile_agg(vec![1], Some(2), 0.5); let root = QueryExpr::BinaryOp { @@ -911,7 +911,7 @@ mod tests { rhs: Rc::new(agg), vector_match: None, }; - let shared = share_common_subtrees(vec![("q", root)]); + let shared = share_common_sub_dags(vec![("q", root)]); let [(_, root)] = shared.as_slice() else { panic!("expected 1 root"); }; @@ -945,12 +945,12 @@ mod tests { } #[test] - fn structural_hash_of_an_internally_shared_tree_matches_the_unshared_equivalent() { + fn structural_hash_of_an_internally_shared_dag_matches_the_unshared_equivalent() { // The same BinaryOp-with-shared-branches shape as - // `dag_node_count_deduplicates_an_internally_shared_subtree` below: + // `dag_node_count_deduplicates_an_internally_shared_sub_dag` below: // hashing it (however the memoization internally short-circuits the // second branch) must produce the exact same value as hashing a - // structurally-identical tree built with *no* sharing at all — the + // structurally-identical DAG built with *no* sharing at all — the // whole point of memoization is not changing the answer, only the // work needed to reach it. let agg = quantile_agg(vec![1], Some(2), 0.5); @@ -1012,12 +1012,12 @@ mod tests { } #[test] - fn dag_node_count_deduplicates_an_internally_shared_subtree() { - // Same shape as `single_query_shares_its_own_repeated_subtree`: a + fn dag_node_count_deduplicates_an_internally_shared_sub_dag() { + // Same shape as `single_query_shares_its_own_repeated_sub_dag`: a // BinaryOp whose two branches are the *same* Rc after - // `share_common_subtrees` (2 nodes: Scan + Aggregate) — the root + // `share_common_sub_dags` (2 nodes: Scan + Aggregate) — the root // itself makes 3 unique nodes total (BinaryOp, Aggregate, Scan), - // not 5 (which a tree-walk / naive serialization, counting the + // not 5 (which a per-path walk / naive serialization, counting the // shared branch's 2 nodes twice, would report). let agg = quantile_agg(vec![1], Some(2), 0.5); let root = QueryExpr::BinaryOp { @@ -1026,7 +1026,7 @@ mod tests { rhs: Rc::new(agg), vector_match: None, }; - let shared = share_common_subtrees(vec![("q", root)]); + let shared = share_common_sub_dags(vec![("q", root)]); let [(_, root)] = shared.as_slice() else { panic!("expected 1 root"); }; @@ -1042,18 +1042,18 @@ mod tests { #[test] fn dag_node_count_deduplicates_across_two_workload_roots() { // Two workload roots sharing one Aggregate after - // `share_common_subtrees` (the `duplicate_workload_queries_...` + // `share_common_sub_dags` (the `duplicate_workload_queries_...` // shape from `crates/integration-tests/tests/cse.rs`, built // directly here): each root's own `dag_node_count` must report the - // shared subtree's real size once, not double-count anything — + // shared sub-DAG's real size once, not double-count anything — // there's nothing *to* double-count from a single root's own count // in this case (no root references the shared node twice), so this // pins the simpler, more common case that a per-candidate cost - // proxy (`CseCandidate::subtree` in `asap-aware-mapping`) actually + // proxy (`CseCandidate::sub-DAG` in `asap-aware-mapping`) actually // exercises: counting one occurrence's own reachable DAG size. let a = quantile_agg(vec![1], Some(2), 0.5); let b = quantile_agg(vec![1], Some(2), 0.5); - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); + let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; @@ -1065,7 +1065,7 @@ mod tests { #[test] fn dedup_gates_sharing_the_same_as_aggregate() { // `Dedup { cols }` adds `cols` as a unique key — so two identical - // `Dedup` subtrees over a keyed column *do* merge, exercising the + // `Dedup` sub-DAGs over a keyed column *do* merge, exercising the // legality gate on a non-`Aggregate` node. let dedup = |cols: Vec| QueryExpr::Dedup { cols, @@ -1073,7 +1073,7 @@ mod tests { }; let a = dedup(vec![1]); let b = dedup(vec![1]); - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); + let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; @@ -1102,7 +1102,7 @@ mod tests { let a = without_agg(); let b = without_agg(); assert!(!a.output_schema().unwrap().has_unique_key()); - let shared = share_common_subtrees(vec![("a", a), ("b", b)]); + let shared = share_common_sub_dags(vec![("a", a), ("b", b)]); let [(_, ra), (_, rb)] = shared.as_slice() else { panic!("expected 2 roots"); }; diff --git a/crates/types/src/pre_asap/expr_ir.rs b/crates/types/src/pre_asap/expr_ir.rs index 2aed64aa2..cc201a617 100644 --- a/crates/types/src/pre_asap/expr_ir.rs +++ b/crates/types/src/pre_asap/expr_ir.rs @@ -1,11 +1,11 @@ //! Column-reference and scalar-operator vocabulary shared by the whole -//! canonical [`QueryExpr`](super::query_expr::QueryExpr) tree. +//! canonical [`QueryExpr`](super::query_expr::QueryExpr) DAG. //! //! Issue #205: the scalar expression shapes (`Column`/`Literal`/`Compare`/…) -//! used to live in a separate, self-recursive `Expr` tree here, reachable +//! used to live in a separate, self-recursive `Expr` DAG here, reachable //! from `QueryExpr` only through wrapper fields (`Predicate`, `ProjectItem`, //! `SortKey`). They're variants of `QueryExpr` itself now — one recursive -//! tree, not two type families joined by wrappers — generic over the same +//! DAG, not two type families joined by wrappers — generic over the same //! column-reference state `C` the rest of `QueryExpr` already carries //! (issue #179): [`ColumnRef`] (name-based, front-end-emitted) or //! [`ColumnId`](super::schema::ColumnId) (positional, once bound). diff --git a/crates/types/src/pre_asap/mod.rs b/crates/types/src/pre_asap/mod.rs index f8f7e3519..bb9b8e309 100644 --- a/crates/types/src/pre_asap/mod.rs +++ b/crates/types/src/pre_asap/mod.rs @@ -1,7 +1,7 @@ //! The canonical pre-ASAP intent algebra IR. //! //! - [`query_expr`] — the canonical, language- and deployment-independent -//! intent algebra: one recursive [`QueryExpr`] tree (relational operators +//! intent algebra: one recursive [`QueryExpr`] DAG (relational operators //! *and* scalar expression shapes both, since issue #205) + [`AggIntent`], //! generic over the column-reference state (positional [`ColumnId`] once //! bound, name-based [`ColumnRef`] before). @@ -12,16 +12,16 @@ //! - [`schema`] — the per-edge [`Schema`] every node carries. //! - [`schema_resolver`] / [`column_resolution`] — name resolution: turn a `ColumnRef` //! into a positional `ColumnId` against an in-scope [`Schema`]. -//! - [`resolve`] — binds a whole front-end-emitted [`UnresolvedQueryExpr`] tree to +//! - [`resolve`] — binds a whole front-end-emitted [`UnresolvedQueryExpr`] DAG to //! canonical [`ResolvedQueryExpr`] (issue #179): both front ends //! (`asap-frontend-promql`, `asap-frontend-sql`) construct `UnresolvedQueryExpr` //! directly during their own `interpret` step and call //! [`resolve_root`] on the result — there is no separate per-language -//! relational tree or converter anymore. +//! relational DAG or converter anymore. //! - [`canonicalize`] — post-lowering structural normalization of [`QueryExpr`] //! (issue #34), run by [`resolve_root`]. //! - [`cse`] — workload-level structural common-subexpression elimination -//! over an already-`resolve_root`'d tree (issue #212, #222, #223), run +//! over an already-`resolve_root`'d DAG (issue #212, #222, #223), run //! *after* `resolve_root` / `canonicalize` and *before* implementation //! (`asap_aware_mapping::replacement`). //! @@ -50,7 +50,7 @@ pub use column_resolution::{ output_schema_for_aggregate, resolve_column_ref, resolve_column_refs, resolve_expr, ResolveError, }; -pub use cse::share_common_subtrees; +pub use cse::share_common_sub_dags; pub use expr_ir::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; pub use query_expr::{ aggregate_output_schema, any_measure_filtered, AtModifier, BinaryOpKind, ColState, DataModel, @@ -59,6 +59,6 @@ pub use query_expr::{ SortKey, Source, TimeShift, UnresolvedQueryExpr, VectorGrouping, VectorMatch, VectorMatchKind, WindowFrame, WindowFrameBound, WindowFrameOffset, WindowFrameUnits, WindowFuncKind, }; -pub use resolve::{resolve_root, ResolveTreeError}; +pub use resolve::{resolve_root, ResolveDAGError}; pub use schema::{Column, ColumnId, DataType, Schema}; pub use schema_resolver::{SchemaCatalog, SchemaResolver, UsageDerivedCatalog}; diff --git a/crates/types/src/pre_asap/query_expr.rs b/crates/types/src/pre_asap/query_expr.rs index 9f0e800c1..60f0e4ec0 100644 --- a/crates/types/src/pre_asap/query_expr.rs +++ b/crates/types/src/pre_asap/query_expr.rs @@ -1,13 +1,13 @@ //! The canonical pre-ASAP intent algebra IR. //! -//! Language- and deployment-independent. `Rc`-owned tree — a child field is +//! Language- and deployment-independent. `Rc`-owned DAG — a child field is //! `Rc>` rather than `Box>` so a structurally //! identical sub-expression can be shared (the same `Rc`) across more than //! one parent, within one query or across a `QueryWorkload` batch, instead of //! being duplicated. Nothing in this module produces that sharing on its //! own — construction still allocates a fresh `Rc` per node, the same shape -//! as the old `Box` tree — a separate CSE pass is what turns two -//! independently constructed, structurally-equal subtrees into two +//! as the old `Box` DAG — a separate CSE pass is what turns two +//! independently constructed, structurally-equal sub-DAGs into two //! references to one `Rc` (issue #212, #222). Column identity is //! **positional** (`Aggregate.reduction: Reduction`, wrapping `GroupKeys` //! for the grouped case), resolved by the [`SchemaResolver`](super::schema_resolver) against @@ -23,13 +23,13 @@ use super::agg_intent::AggIntent; use super::expr_ir::{ArithmeticOpKind, ColumnRef, CompareOpKind, ScalarValue}; use super::schema::{Column, ColumnId, DataType, Schema}; -/// The column-reference resolution state a [`QueryExpr`] tree carries — +/// The column-reference resolution state a [`QueryExpr`] DAG carries — /// [`ColumnId`] (the default, and what the bare `QueryExpr` name has always /// meant) once the [`SchemaResolver`](super::schema_resolver::SchemaResolver) has resolved every /// reference positionally, or the front-end-emitted, name-based [`ColumnRef`] /// before binding. The only place the two states differ in *shape* rather /// than just in which type fills `C` is [`QueryExpr::Scan`]'s `schema` field: -/// a bound tree's binding schema is always known (the SchemaResolver is total, so +/// a bound DAG's binding schema is always known (the SchemaResolver is total, so /// [`ScanSchema`](Self::ScanSchema) `= Schema`); an unresolved front-end /// `Scan` knows its schema only when the front end already has it without /// binding — a SQL leaf, catalog-backed (`Some`) — `None` (PromQL) defers to @@ -37,7 +37,7 @@ use super::schema::{Column, ColumnId, DataType, Schema}; pub trait ColState: Clone + std::fmt::Debug + PartialEq + Serialize + for<'de> Deserialize<'de> { - /// What [`QueryExpr::Scan`]'s `schema` field holds for a tree in this state. + /// What [`QueryExpr::Scan`]'s `schema` field holds for a DAG in this state. type ScanSchema: Clone + std::fmt::Debug + PartialEq + Serialize + for<'de> Deserialize<'de>; } @@ -49,7 +49,7 @@ impl ColState for ColumnRef { type ScanSchema = Option; } -/// Errors from schema derivation over a canonical tree. +/// Errors from schema derivation over a canonical DAG. #[derive(Debug, Error)] pub enum QueryExprError { #[error("invalid scalar function signature: {0}")] @@ -652,7 +652,7 @@ impl ConcatDiscriminatorKey { #[serde(bound(serialize = "C: ColState", deserialize = "C: ColState"))] pub enum QueryExpr { /// Outermost leaf. `schema` is the **binding schema** — the resolved column - /// set every positional `ColumnId` in the tree indexes into, *not* a full + /// set every positional `ColumnId` in the DAG indexes into, *not* a full /// description of the runtime row — once bound (`schema: Schema`, always /// present: the [`SchemaResolver`](super::schema_resolver) is total). Before binding, a /// front-end-emitted `Scan` (`C = ColumnRef`) knows it only when the front @@ -671,7 +671,7 @@ pub enum QueryExpr { predicates: Vec>, schema: C::ScanSchema, }, - /// A scalar sub-expression sitting in an **operator-tree position** — a + /// A scalar sub-expression sitting in an **operator-DAG position** — a /// [`BinaryOp`](Self::BinaryOp) operand for ` op ` /// thresholds / unit conversions (#35), a /// [`PromqlVectorFromScalar`](Self::PromqlVectorFromScalar) child, or a @@ -680,7 +680,7 @@ pub enum QueryExpr { /// Formerly its own leaf variant, `PromqlScalar(f64)`. Issue #220: that /// variant held exactly the same value [`Literal`](Self::Literal) does /// (every PromQL scalar is `f64`), duplicating it for no reason but - /// *which tree position* it was allowed to appear in. This wrapper + /// *which DAG position* it was allowed to appear in. This wrapper /// carries that position instead of the value — the inner node is an /// ordinary scalar sub-language expression (in practice always /// `Literal(ScalarValue::Float64(_))`, since a front end only ever @@ -941,9 +941,9 @@ pub enum QueryExpr { // ── Scalar expression shapes (issue #205) ─────────────────────────── // - // Formerly a separate, self-recursive `Expr` tree, reachable from the + // Formerly a separate, self-recursive `Expr` DAG, reachable from the // operator variants above only through wrapper fields (`Predicate`, - // `ProjectItem`, `SortKey`). They're variants of this same tree now — a + // `ProjectItem`, `SortKey`). They're variants of this same DAG now — a // scalar sub-expression is only ever reachable through one of those same // wrapper positions (`Filter.pred`, `ProjectItem.expr`, `Aggregate.having`, // `PromqlRelabel.value`, `SQLWindowFunc.args`, …), which is a *convention* this @@ -957,7 +957,7 @@ pub enum QueryExpr { // restricting which variants are constructible in a scalar position) adds // real type-level machinery for a distinction every constructor already // has to get right structurally anyway (a `Filter` is never built with an - // operator subtree as its `pred`). + // operator sub-DAG as its `pred`). /// A column reference — unresolved [`ColumnRef`] (front-end-emitted, `C = /// ColumnRef`) or positional [`ColumnId`] (once bound, `C = ColumnId`). Column(C), @@ -1014,7 +1014,7 @@ pub enum QueryExpr { impl QueryExpr { /// Construct the [`PromqlScalarBridge`](Self::PromqlScalarBridge) leaf /// for a bare PromQL numeric literal / folded constant scalar (issue - /// #220) — `Literal(ScalarValue::Float64(v))` at an operator-tree + /// #220) — `Literal(ScalarValue::Float64(v))` at an operator-DAG /// position. The one constructor every front end / test that used to /// write `QueryExpr::PromqlScalar(v)` should use instead. pub fn promql_scalar(v: f64) -> Self { @@ -1086,7 +1086,7 @@ impl QueryExpr { } } - /// Recursively collect every column reference in a **scalar** subtree — + /// Recursively collect every column reference in a **scalar** sub-DAG — /// used by the [`SchemaResolver`](super::schema_resolver::SchemaResolver) to seed usage-derived /// leaf schemas, and available to post-ASAP binding for column-lineage / /// selectivity. @@ -1146,20 +1146,20 @@ impl QueryExpr { } } -/// The canonical, positional, resolved tree — what the bare `QueryExpr` name +/// The canonical, positional, resolved DAG — what the bare `QueryExpr` name /// has always meant (the default `C = ColumnId`). Every existing consumer /// keeps using `QueryExpr` unparameterized; this alias exists only to name /// the resolved state explicitly at a use site that also wants to name /// [`UnresolvedQueryExpr`] nearby. pub type ResolvedQueryExpr = QueryExpr; -/// The front-end-emitted, name-based, unresolved tree — +/// The front-end-emitted, name-based, unresolved DAG — /// `QueryExpr`: front ends construct this directly during their /// own `interpret` step (issue #179), and the [`SchemaResolver`](super::schema_resolver) /// resolves it into [`ResolvedQueryExpr`]. pub type UnresolvedQueryExpr = QueryExpr; -// `output_schema` needs a fully bound tree — it reads `Scan.schema` as a plain +// `output_schema` needs a fully bound DAG — it reads `Scan.schema` as a plain // `Schema` and resolves every scalar `Expr::Column` positionally — so it lives // only on the resolved instantiation, not `impl QueryExpr`. // Same reasoning as `AggIntent`'s `output_column`/`requires`/`is_per_series` @@ -1171,7 +1171,7 @@ impl QueryExpr { infer_expr_type(self, input) } - /// Output schema of the root of a canonical tree. + /// Output schema of the root of a canonical DAG. pub fn output_schema(&self) -> Result { match self { QueryExpr::Scan { schema, .. } => Ok(schema.clone()), @@ -2629,7 +2629,7 @@ mod tests { /// place of the old `PromqlScalar(v)` leaf — wraps exactly /// `Literal(ScalarValue::Float64(v))`: the same value a SQL-emitted typed /// float literal in a scalar-sub-language position would carry, just at a - /// different tree position. `as_promql_scalar` is the round-trip inverse. + /// different DAG position. `as_promql_scalar` is the round-trip inverse. #[test] fn promql_scalar_bridges_a_literal_float_at_an_operator_position() { let bridge = QueryExpr::::promql_scalar(2.5); @@ -2641,7 +2641,7 @@ mod tests { // The same value a SQL `Compare`/`Arithmetic` operand would carry, in // its native (unwrapped, no row schema) scalar-sub-language position — - // no longer a different variant, just not bridged to this tree + // no longer a different variant, just not bridged to this DAG // position. let sql_literal = QueryExpr::::Literal(ScalarValue::Float64(2.5)); assert_eq!(bridge.as_promql_scalar(), Some(2.5)); @@ -2655,9 +2655,9 @@ mod tests { assert_eq!(scan(vec![], None, vec![]).as_promql_scalar(), None); } - /// Pins the tree-position distinction issue #220 asks for: the very same + /// Pins the DAG-position distinction issue #220 asks for: the very same /// `Literal(ScalarValue::Float64(_))` value has a row schema when it sits - /// at the operator-tree position (wrapped in `PromqlScalarBridge` — a + /// at the operator-DAG position (wrapped in `PromqlScalarBridge` — a /// `BinaryOp` operand, `PromqlVectorFromScalar` child, or a query root), /// and has none when it sits bare, in a scalar-sub-language position /// (`Compare`/`Arithmetic`/… operand) — no longer decided by which of two diff --git a/crates/types/src/pre_asap/resolve.rs b/crates/types/src/pre_asap/resolve.rs index 95b482e54..104c51a68 100644 --- a/crates/types/src/pre_asap/resolve.rs +++ b/crates/types/src/pre_asap/resolve.rs @@ -20,7 +20,7 @@ //! through logical optimization, only going positional once they lower to a //! physical plan. Resolving once, immediately after each front end's own //! `interpret` step, is the better trade for *this* codebase's shape — one -//! front-end-facing tree feeding several independent downstream passes +//! front-end-facing DAG feeding several independent downstream passes //! (`canonicalize`, the cost model, `dag_export`, schema/type inference, //! `asap-aware-mapping`'s summary binding) — for three concrete reasons: //! @@ -30,7 +30,7 @@ //! schemas are concatenated. A bare name is ambiguous the moment two sources //! share one; `ColumnId` is what makes "the second `service`, position 4, not //! the first" a fact recorded once, instead of a lookup redone at every use site. -//! 2. **A name's meaning changes going up the tree.** `Project` renames/aliases, +//! 2. **A name's meaning changes going up the DAG.** `Project` renames/aliases, //! `Aggregate` collapses columns and introduces synthetic ones, `Join` //! concatenates two schemas — a name valid at a `Scan` leaf isn't //! automatically the right binding three nodes up; it has to be reinterpreted @@ -64,9 +64,9 @@ use super::query_expr::{ use super::schema::{ColumnId, Schema}; use super::schema_resolver::SchemaResolver; -/// Errors from resolving a canonical, unresolved [`UnresolvedQueryExpr`] tree. +/// Errors from resolving a canonical, unresolved [`UnresolvedQueryExpr`] DAG. #[derive(Debug, Error)] -pub enum ResolveTreeError { +pub enum ResolveDAGError { /// A column reference did not resolve against its in-scope schema. #[error("column resolution failed: {0}")] Resolve(#[from] ResolveError), @@ -76,22 +76,22 @@ pub enum ResolveTreeError { Schema(#[from] QueryExprError), } -/// Resolve a whole [`UnresolvedQueryExpr`] tree rooted at `tree` into canonical +/// Resolve a whole [`UnresolvedQueryExpr`] DAG rooted at `dag` into canonical /// [`ResolvedQueryExpr`]: binds every `ColumnRef` to a `ColumnId` via the /// [`SchemaResolver`], then [`canonicalize`](super::canonicalize::canonicalize)s the /// result. -pub fn resolve_root(tree: &UnresolvedQueryExpr) -> Result { - resolve_root_with_inherited(tree, &[]) +pub fn resolve_root(dag: &UnresolvedQueryExpr) -> Result { + resolve_root_with_inherited(dag, &[]) } /// [`resolve_root`] with label names inherited from an enclosing scope seeded /// into the leaf schema, used when re-binding a `BinaryOp` side (issue #52). fn resolve_root_with_inherited( - tree: &UnresolvedQueryExpr, + dag: &UnresolvedQueryExpr, inherited: &[String], -) -> Result { - let fallback = SchemaResolver::new().resolve_schema_with_inherited(tree, inherited); - let l3 = resolve(tree, &fallback)?; +) -> Result { + let fallback = SchemaResolver::new().resolve_schema_with_inherited(dag, inherited); + let l3 = resolve(dag, &fallback)?; Ok(super::canonicalize::canonicalize(l3)) } @@ -100,11 +100,11 @@ fn resolve_root_with_inherited( /// derived output schema — so a `JOIN`'s concatenated schema and a cross- /// series aggregate's frozen-closed output bind to the right positions. fn resolve( - tree: &UnresolvedQueryExpr, + dag: &UnresolvedQueryExpr, fallback: &Schema, -) -> Result { +) -> Result { use super::query_expr::QueryExpr as QE; - Ok(match tree { + Ok(match dag { QE::Scan { source, predicates, @@ -123,7 +123,7 @@ fn resolve( } // `PromqlScalarBridge`'s child is a scalar-sub-language node (issue - // #220) sitting at this operator-tree position — resolved through + // #220) sitting at this operator-DAG position — resolved through // `resolve_expr`, same as every other scalar position (`Predicate`, // `ProjectItem.expr`, …), not the operator walk. In practice it's // always a `Literal`, which has no `ColumnRef` to resolve, so @@ -232,7 +232,7 @@ fn resolve( }; let having = having .as_ref() - .map(|Predicate(h)| -> Result { + .map(|Predicate(h)| -> Result { let out_schema = aggregate_output_schema( &child_schema, &reduction, @@ -279,7 +279,7 @@ fn resolve( // other side. let discriminator_unique_key = discriminator_unique_key .as_ref() - .map(|key| -> Result<_, ResolveTreeError> { + .map(|key| -> Result<_, ResolveDAGError> { let schema = children .first() .ok_or(QueryExprError::EmptyConcat)? @@ -427,7 +427,7 @@ fn resolve( // different label sets, so each branch resolves against its OWN // bound schema; but an independently-bound side still has to see // label names an *enclosing* node references (issue #52). - let own = super::schema_resolver::collect_referenced_columns(tree); + let own = super::schema_resolver::collect_referenced_columns(dag); let inherited: Vec = inherited_names(fallback) .into_iter() .filter(|n| !own.contains(n)) @@ -501,7 +501,7 @@ fn resolve_group_keys( /// uniformly to every `Aggregate`, not just PromQL's: SQL's `GROUP BY` keys /// are always genuinely present (DataFusion validates the plan), so the /// "drop instead of reject" branch is simply never exercised there — the -/// lenient resolver is a no-op difference for a SQL tree, not a behavior +/// lenient resolver is a no-op difference for a SQL DAG, not a behavior /// change. fn resolve_reduction( reduction: &Reduction, @@ -815,7 +815,7 @@ mod tests { } /// Issue #228 review, end-to-end: `resolve_root` over a `Concat` whose - /// discriminator column is referenced *nowhere else* in the tree, with a + /// discriminator column is referenced *nowhere else* in the DAG, with a /// schema-less (usage-derived) leaf `Scan` in the first branch — exactly /// the scenario the review flagged. Before the `schema_resolver.rs` fix, the /// SchemaResolver's fallback schema wouldn't contain `phi` at all, and this diff --git a/crates/types/src/pre_asap/schema_resolver.rs b/crates/types/src/pre_asap/schema_resolver.rs index b8d23b6a3..d6bbd1610 100644 --- a/crates/types/src/pre_asap/schema_resolver.rs +++ b/crates/types/src/pre_asap/schema_resolver.rs @@ -1,7 +1,7 @@ //! The **SchemaResolver** — name resolution as an explicit pass. //! //! [`SchemaResolver::resolve_schema`] produces the complete, self-contained [`Schema`] every -//! `ColumnId` in the canonical tree indexes into. [`resolve`](super::resolve) +//! `ColumnId` in the canonical DAG indexes into. [`resolve`](super::resolve) //! then becomes purely structural: it threads the SchemaResolver's schema and //! positional resolution downstream is **total**. //! @@ -66,27 +66,27 @@ impl SchemaResolver { Self { catalog } } - /// Resolve the complete [`Schema`] in scope for a query rooted at `tree`. + /// Resolve the complete [`Schema`] in scope for a query rooted at `dag`. /// /// Contains the time axis, the synthetic `value` column, and one column - /// per distinct name referenced anywhere in the tree — so positional + /// per distinct name referenced anywhere in the DAG — so positional /// `ColumnId` resolution downstream is total. - pub fn resolve_schema(&self, tree: &UnresolvedQueryExpr) -> Schema { - self.resolve_schema_with_inherited(tree, &[]) + pub fn resolve_schema(&self, dag: &UnresolvedQueryExpr) -> Schema { + self.resolve_schema_with_inherited(dag, &[]) } /// Like [`resolve_schema`](Self::resolve_schema), but also seeds `inherited` label names that are - /// referenced by an **enclosing** scope rather than by `tree` itself. This is + /// referenced by an **enclosing** scope rather than by `dag` itself. This is /// how an independently-bound `BinaryOp` side (each side re-binds against its - /// own sub-tree) still sees an outer aggregate's group keys — e.g. the + /// own sub-DAG) still sees an outer aggregate's group keys — e.g. the /// `__name__` / `job` in `sum by (__name__)(a or b)`, which appear in neither /// side's own matchers (issue #52). pub fn resolve_schema_with_inherited( &self, - tree: &UnresolvedQueryExpr, + dag: &UnresolvedQueryExpr, inherited: &[String], ) -> Schema { - let mut columns: Vec = leftmost_scan_name(tree) + let mut columns: Vec = leftmost_scan_name(dag) .and_then(|name| self.catalog.columns_for(name)) .unwrap_or_else(default_leaf_columns); @@ -99,7 +99,7 @@ impl SchemaResolver { // Append one column per referenced-but-unknown name (group keys etc.), // plus any inherited-from-enclosing-scope names. - let referenced = collect_referenced_columns(tree); + let referenced = collect_referenced_columns(dag); for name in referenced.iter().chain(inherited) { if !columns.iter().any(|c| c.name == *name) { columns.push(Column::new(name.clone(), DataType::Utf8, true)); @@ -136,13 +136,13 @@ fn push_ref_name(c: &ColumnRef, out: &mut Vec) { } } -/// The leftmost `Scan`'s source name in a canonical (`UnresolvedQueryExpr`) tree — +/// The leftmost `Scan`'s source name in a canonical (`UnresolvedQueryExpr`) DAG — /// the [`collect_referenced_columns`] counterpart to what a dedicated -/// `Source` leaf type would carry as a method; the canonical tree's `Scan` +/// `Source` leaf type would carry as a method; the canonical DAG's `Scan` /// leaf needs this walk written out instead. -fn leftmost_scan_name(tree: &UnresolvedQueryExpr) -> Option<&str> { +fn leftmost_scan_name(dag: &UnresolvedQueryExpr) -> Option<&str> { use UnresolvedQueryExpr as QE; - match tree { + match dag { QE::Scan { source, .. } => Some(match source { super::query_expr::Source::TimeSeries { metric } => metric.as_str(), super::query_expr::Source::Table { table_ref } => table_ref.as_str(), @@ -191,7 +191,7 @@ fn leftmost_scan_name(tree: &UnresolvedQueryExpr) -> Option<&str> { } } -/// Collect every distinct column name referenced anywhere in `tree` that +/// Collect every distinct column name referenced anywhere in `dag` that /// resolves positionally — every place a front end constructing /// [`QueryExpr`](super::query_expr::QueryExpr) directly (issue /// #179) puts a name-based reference: `Scan.predicates`, `Aggregate`'s @@ -200,7 +200,7 @@ fn leftmost_scan_name(tree: &UnresolvedQueryExpr) -> Option<&str> { /// `SQLWindowFunc.args`/`partition_by`/`order_by`, `Join.pred`, `PromqlRelabel.value`. /// The SchemaResolver seeds these into the usage-derived leaf so positional /// resolution downstream is total. -pub(crate) fn collect_referenced_columns(tree: &UnresolvedQueryExpr) -> Vec { +pub(crate) fn collect_referenced_columns(dag: &UnresolvedQueryExpr) -> Vec { use UnresolvedQueryExpr as QE; fn named(expr: &UnresolvedQueryExpr, out: &mut Vec) { for c in expr.columns_referenced() { @@ -322,7 +322,7 @@ pub(crate) fn collect_referenced_columns(tree: &UnresolvedQueryExpr) -> Vec Vec = Vec::new(); - walk(tree, &mut out); + walk(dag, &mut out); out.sort(); out.dedup(); out @@ -390,7 +390,7 @@ mod tests { #[test] fn pearson_corr_inputs_seed_usage_derived_schema() { use crate::pre_asap::{AggIntent, Reduction}; - let tree = UnresolvedQueryExpr::Aggregate { + let dag = UnresolvedQueryExpr::Aggregate { reduction: Reduction::by(vec![]), measures: vec![AggIntent::PearsonCorr { left: ColumnRef::Named("x".into()), @@ -401,8 +401,8 @@ mod tests { having: None, child: Rc::new(src("m")), }; - assert_eq!(collect_referenced_columns(&tree), vec!["x", "y"]); - let schema = SchemaResolver::new().resolve_schema(&tree); + assert_eq!(collect_referenced_columns(&dag), vec!["x", "y"]); + let schema = SchemaResolver::new().resolve_schema(&dag); assert!(schema.column_id("x").is_some()); assert!(schema.column_id("y").is_some()); } @@ -420,7 +420,7 @@ mod tests { fn sort_partition_keys_land_in_schema() { // Per-group ranking keys (`topk by (host)` → `Sort.partition_by`) must be // seeded into the usage-derived leaf so they resolve positionally. - let tree = UnresolvedQueryExpr::Sort { + let dag = UnresolvedQueryExpr::Sort { keys: vec![super::super::query_expr::SortKey { expr: UnresolvedQueryExpr::Column(ColumnRef::SampleValue), ascending: false, @@ -429,23 +429,23 @@ mod tests { partition_by: GroupKeys::by(vec![ColumnRef::Named("host".into())]), child: Rc::new(src("hits")), }; - let schema = SchemaResolver::new().resolve_schema(&tree); + let schema = SchemaResolver::new().resolve_schema(&dag); assert!(schema.column_id("host").is_some()); } /// Issue #228 review: a `Concat`'s `discriminator_unique_key` columns — - /// even one referenced nowhere else in the tree — must be seeded into + /// even one referenced nowhere else in the DAG — must be seeded into /// the usage-derived fallback schema, exactly like `Dedup.cols`, or /// `resolve.rs`'s later `resolve_column_ref` fails `NotFound` for a /// column the caller correctly named. #[test] fn concat_discriminator_key_is_seeded_into_the_resolver_schema() { - let tree = UnresolvedQueryExpr::concat_with_discriminator( + let dag = UnresolvedQueryExpr::concat_with_discriminator( vec![src("m")], ColumnRef::Named("phi".into()), vec![ColumnRef::Named("host".into())], ); - let schema = SchemaResolver::new().resolve_schema(&tree); + let schema = SchemaResolver::new().resolve_schema(&dag); assert!( schema.column_id("phi").is_some(), "discriminator column must be seeded" @@ -458,7 +458,7 @@ mod tests { #[test] fn inherited_names_are_seeded_alongside_referenced() { - // A `BinaryOp` side re-binds against its own sub-tree, but must still see + // A `BinaryOp` side re-binds against its own sub-DAG, but must still see // an enclosing aggregate's group key (`__name__` / `job`) that appears in // neither side's own matchers (issue #52). `resolve_schema_with_inherited` seeds it. let schema = diff --git a/crates/types/tests/planner_vocabulary.rs b/crates/types/tests/planner_vocabulary.rs index 567e2c284..f14a56ee8 100644 --- a/crates/types/tests/planner_vocabulary.rs +++ b/crates/types/tests/planner_vocabulary.rs @@ -33,14 +33,14 @@ fn window_edge_names_preserve_wire_values() { // External consumers can use the new resolver and resource names without changing behavior. #[test] fn renamed_schema_and_handoff_apis_are_public() { - let tree = UnresolvedQueryExpr::Scan { + let dag = UnresolvedQueryExpr::Scan { source: Source::TimeSeries { metric: "requests".into(), }, predicates: vec![], schema: None, }; - let schema = SchemaResolver::new().resolve_schema(&tree); + let schema = SchemaResolver::new().resolve_schema(&dag); assert!(schema.column_id("value").is_some()); let bytes = PhysicalHandoffBytes { network_bytes: 12, diff --git a/docs/design_docs/architecture/parse-and-canonicalize.md b/docs/design_docs/architecture/parse-and-canonicalize.md index b111f9730..61149efbb 100644 --- a/docs/design_docs/architecture/parse-and-canonicalize.md +++ b/docs/design_docs/architecture/parse-and-canonicalize.md @@ -18,7 +18,7 @@ SQL ORDER BY COUNT(*) DESC LIMIT 10 ## Canonicalize -`canonicalize` normalizes semantically equivalent intent trees so that equivalent queries +`canonicalize` normalizes semantically equivalent intent DAGs so that equivalent queries from different languages, or differently phrased queries within one language, converge on the same canonical shape. diff --git a/docs/design_docs/architecture/physical-plan-integration.md b/docs/design_docs/architecture/physical-plan-integration.md index 47fc3a1b6..7fffc9fdf 100644 --- a/docs/design_docs/architecture/physical-plan-integration.md +++ b/docs/design_docs/architecture/physical-plan-integration.md @@ -128,7 +128,7 @@ select each concrete implementation and provide all edges, resource facts, multiplicities, source ownership, and stable physical identities. The planner fails closed when any reachable `SummaryExpr` node lacks that binding. -The raw/query portion of a streaming comparison remains a `PhysicalDag` using +The raw/query portion of a streaming comparison remains a `PhysicalDAG` using the canonical `PhysicalOperator` and `OperatorStatistics` pairing. Summary evidence is kept separate only where lifecycle-driven update, retention, and expiration multiplicities require facts beyond the query-DAG @@ -425,7 +425,7 @@ requires a normal result; setting both guards or attaching a guard to a non-divi operator is invalid. Compilers must preserve this typed condition rather than recovering average semantics from query text. -### Candidate pruning is a subgraph +### Candidate pruning is a sub-DAG Candidate-based TopK uses a summary key readout, a general semi-join over explicit matching key columns, grouped Sort by the authoritative score, and @@ -493,7 +493,7 @@ workload schema cannot identify separate backlog and arrival populations. The adapter fails explicitly rather than guessing a split. The estimator version is `summary-maintenance-resource-v2`; evidence type names drop the `Streaming` prefix (`SummaryMaintenanceInputs`, `SummaryPhysicalInputEvidence`, `SummaryAggregateEvidence`, -`RetainedSubDagEvidence`, `RawInputEvidence`, and the summary window/alternative +`RetainedSubDAGEvidence`, `RawInputEvidence`, and the summary window/alternative types). Update source imports; no legacy-name aliases are provided. Regressions cover a fixed snapshot with no rate evidence, contradictory arrival diff --git a/docs/design_docs/concepts/post-asap-ir.md b/docs/design_docs/concepts/post-asap-ir.md index 67b5535a1..6c9aa1461 100644 --- a/docs/design_docs/concepts/post-asap-ir.md +++ b/docs/design_docs/concepts/post-asap-ir.md @@ -2,7 +2,7 @@ The goal of the post-ASAP IR is to represent operations using ASAP primitives such as sketches, exact summaries, samples and wavelets. Post-ASAP IR also -retains exact Pre-ASAP subtrees and supports operations over summary readouts, +retains exact Pre-ASAP sub-DAGs and supports operations over summary readouts, since only some query operations can be satisfied using summaries. The lists below cover every current variant of @@ -36,7 +36,7 @@ summary family supports incremental maintenance. ## Exact work and composition nodes -- `KeepPreAsap`: retain an exact Pre-ASAP subtree when it is not rewritten. +- `KeepPreAsap`: retain an exact Pre-ASAP sub-DAG when it is not rewritten. - `BinaryOp`: combine independently planned operands with the specified binary semantics and execution timing. - `ValueOperation`: apply aggregate, exact-function, population, projection, @@ -55,15 +55,15 @@ approximate readouts still require composed accuracy guarantees. See the and [physical-plan integration](../architecture/physical-plan-integration.md) for the corresponding correctness and realization requirements. -## Tree and exported DAG forms +## In-memory and exported DAG forms The Pre-ASAP DAG and the Post-ASAP DAG are both logical: they describe what is computed, not which physical operators execute it. The Post-ASAP DAG has two -forms of the same content. Planning builds and shares `SummaryNode` trees. -`compile_post_asap_dag` converts a selected tree into a -[`PostAsapDag`](../../../crates/types/src/post_asap/post_asap_dag.rs) with -stable node IDs and typed edges; `PostAsapDagDocument` is its versioned wire -envelope. Physical compilation consumes `PostAsapDag` and produces a separate +forms of the same content. Planning builds and shares `SummaryNode` DAGs. +`compile_post_asap_dag` converts a selected DAG into a +[`PostAsapDAG`](../../../crates/types/src/post_asap/post_asap_dag.rs) with +stable node IDs and typed edges; `PostAsapDAGDocument` is its versioned wire +envelope. Physical compilation consumes `PostAsapDAG` and produces a separate physical DAG. ## Execution phase @@ -74,8 +74,8 @@ one of these phases. Backend capability restrictions are implementation gaps, not definitions of the operator. Every post-ASAP operator payload supports both phase assignments. Phase is -stored on the `PostAsapDag` node, independently of its operator payload. -`PostAsapDag::with_execution_phases` assigns a phase to every node and updates +stored on the `PostAsapDAG` node, independently of its operator payload. +`PostAsapDAG::with_execution_phases` assigns a phase to every node and updates its edges. Ingestion work cannot depend on a future query result. Default semantic realization still proposes an initial layout; it does not restrict which phase an operator may use. Deployments must separately check that diff --git a/docs/design_docs/concepts/pre-asap-ir.md b/docs/design_docs/concepts/pre-asap-ir.md index 227e61e39..735f522f3 100644 --- a/docs/design_docs/concepts/pre-asap-ir.md +++ b/docs/design_docs/concepts/pre-asap-ir.md @@ -31,7 +31,7 @@ Only semantics that affect correctness, summary applicability, or cost become fi ### PromQL-specific -- PromqlScalarBridge — holds a scalar sub-expression at an operator-tree position. +- PromqlScalarBridge — holds a scalar sub-expression at an operator-DAG position. - EvalTimestamp — provides the evaluation timestamp as a scalar. - PromqlVectorFromScalar — promotes a scalar to a label-less instant vector. - PromqlScalarFromVector — collapses a single-series vector to a scalar. diff --git a/docs/design_docs/decisions/concat-unique-keys.md b/docs/design_docs/decisions/concat-unique-keys.md index 778f6061e..9c0148abf 100644 --- a/docs/design_docs/decisions/concat-unique-keys.md +++ b/docs/design_docs/decisions/concat-unique-keys.md @@ -36,7 +36,7 @@ that a discriminator-based unique key would let it drop? ## What was checked Both current `Concat`-constructing call sites, and every consumer of -`Schema::unique_keys` in the tree: +`Schema::unique_keys` in the DAG: - **PromQL `histogram_quantiles`** — [`walk_histogram_quantiles`](../../../crates/frontend-promql/src/promql.rs). @@ -66,8 +66,8 @@ Both current `Concat`-constructing call sites, and every consumer of `sql_lowering.rs`, `dag_export.rs`, `variant_coverage.rs`, netflow/synthetic test fixtures) — none of them builds a fresh `Concat` with a `Dedup` on top that this feature could remove. -- **Every consumer of `Schema::unique_keys`** in the tree, to check for a - cost beyond "a literal `Dedup` node": `pre_asap::cse::share_common_subtrees` +- **Every consumer of `Schema::unique_keys`** in the DAG, to check for a + cost beyond "a literal `Dedup` node": `pre_asap::cse::share_common_sub_dags` (gates CSE producer-sharing on `Schema::has_unique_key()`) and `asap_aware_mapping::rollup::is_legal_rollup_source` (gates rollup-source legality the same way, on an *`Aggregate`'s* own output schema). Neither @@ -88,7 +88,7 @@ Both current `Concat`-constructing call sites, and every consumer of The investigation's conclusion stands: neither `histogram_quantiles` nor `ROLLUP`/`CUBE`/`GROUPING SETS` lowering emits a `Dedup` (or anything playing that role) after its `Concat` today, so there is nothing redundant in the -tree for a discriminator-based override to remove *right now*. On review, +DAG for a discriminator-based override to remove *right now*. On review, the decision was made to build the extension point anyway, ahead of a proven call-site win, rather than wait for one. That is a legitimate call to make differently from the investigation's own recommendation — "no current @@ -114,7 +114,7 @@ for why it's fine to ship unused. overclaimed. - `QueryExpr::concat(children)` — the ordinary constructor (`discriminator_unique_key: None`), meant to replace the bare `QueryExpr::Concat { children }` struct literal - everywhere in the tree so a future field addition doesn't force every call + everywhere in the DAG so a future field addition doesn't force every call site to re-litigate this choice. - `QueryExpr::concat_with_discriminator(children, discriminator, inner_key)` — the override constructor. @@ -127,7 +127,7 @@ for why it's fine to ship unused. branch's own output schema — the same schema `output_schema()` derives the merged shape from — so the feature works correctly end-to-end for a future caller upstream of `resolve_root`, even though no such caller exists yet. -- Every other match/construction site touching `Concat` across the tree +- Every other match/construction site touching `Concat` across the DAG (`canonicalize.rs`, `cse.rs`, `schema_resolver.rs`, `dag_export.rs`, `asap-aware-mapping`'s `replacement.rs`/`explanation.rs`, and every test/tooling AST walker) was mechanically updated to bind or ignore the new @@ -156,9 +156,9 @@ Three independent things hold `discriminator_unique_key: None` as the observable behavior for every caller that doesn't ask for the override: 1. **Every real construction path defaults to `None`.** `QueryExpr::concat` - hardcodes it; every call site in the tree (including both real lowering + hardcodes it; every call site in the DAG (including both real lowering call sites) uses `concat`, not `concat_with_discriminator`, so nothing in - the current tree can produce `Some` at all. + the current DAG can produce `Some` at all. 2. **`output_schema()`'s branch on the field is additive.** The `None` arm is textually the same clear-and-return the code already did — `s.unique_keys.clear(); ... Ok(s)` — with the `Some` branch reached only when the field is populated. This is @@ -236,7 +236,7 @@ accuracy issue. All three are fixed on the same PR: (which *is* walked, `push_ref_name`-style). Concretely: a future `concat_with_discriminator(branches, discriminator_col, inner_key)` call over an open query, where the discriminator column isn't otherwise - referenced anywhere else in the tree, with a schema-less leaf `Scan` in + referenced anywhere else in the DAG, with a schema-less leaf `Scan` in the first branch — the SchemaResolver's fallback schema wouldn't contain the discriminator name, and `resolve.rs`'s later `resolve_column_ref` call would fail `NotFound` for a column the caller correctly named. Fixed: @@ -294,4 +294,4 @@ accuracy issue. All three are fixed on the same PR: - Reopening SQL's rejection of `GROUPING()` so `lower_grouping_sets` has a real discriminator (`__grouping_id`) to assert — a separate design decision. - Any canonicalization rule or CSE/rollup scenario that would actually *read* - a `Concat`'s asserted `unique_keys` for the first time in the current tree. + a `Concat`'s asserted `unique_keys` for the first time in the current DAG. diff --git a/docs/design_docs/decisions/cse-cost-model.md b/docs/design_docs/decisions/cse-cost-model.md index 5689390aa..ed7e54e08 100644 --- a/docs/design_docs/decisions/cse-cost-model.md +++ b/docs/design_docs/decisions/cse-cost-model.md @@ -4,12 +4,12 @@ ## Context -[`asap_types::pre_asap::cse::share_common_subtrees`](../../../crates/types/src/pre_asap/cse.rs) +[`asap_types::pre_asap::cse::share_common_sub_dags`](../../../crates/types/src/pre_asap/cse.rs) (issue #223 stages 1-2, PR #235) already *detects* every structurally-identical, -legally-shareable (`Schema::unique_keys`-gated) subtree and shares it +legally-shareable (`Schema::unique_keys`-gated) sub-DAG and shares it **unconditionally** — there is no cost gate on top of legality. This document decides the framework for stage 4, "wire workload-level CSE credit into -`CostModel`" — turning "these two subtrees are the same computation" into +`CostModel`" — turning "these two sub-DAGs are the same computation" into "and it's actually worth maintaining one shared summary for them." ## The two textbook framings (as posed in #237) @@ -27,7 +27,7 @@ compares two real, overridable cost estimates for every CSE candidate with two or more consumers: - `cse_recompute_cost(candidate) * candidate.consumer_count` — the total cost - of recomputing the subtree independently at every use site. + of recomputing the sub-DAG independently at every use site. - `cse_shared_maintenance_cost(candidate)` — the cost of keeping one shared summary alive and continuously updated for the workload's lifetime. @@ -40,7 +40,7 @@ the way sharing a relational scan is in a textbook OLTP optimizer — it is a sketch/accumulator that (per this crate's stated purpose: *workload*-level planning, not single-query) is typically kept **continuously updated** as new data arrives, for as long as the workload runs, regardless of how often it's -actually read. A structurally-shareable subtree that is cheap to recompute on +actually read. A structurally-shareable sub-DAG that is cheap to recompute on demand, or rarely queried, can cost more to keep alive as a standing shared summary than to just recompute independently at each of its (few, or cheap) use sites. A blanket "always share" rule cannot express that trade-off; a @@ -63,7 +63,7 @@ same way a real cost-based optimizer would. ## Layering constraint -`share_common_subtrees` lives in `asap-types::pre_asap` — a lower layer that +`share_common_sub_dags` lives in `asap-types::pre_asap` — a lower layer that `asap-aware-mapping` (which owns `CostModel`) depends on, never the reverse. Detection therefore cannot consult cost even if it wanted to. This is why stage 1/2's detection stays unconditional (correctly, as a legality-only @@ -74,13 +74,13 @@ gate) and the cost-aware decision is applied downstream, in [`CandidateLogicalASAPDAGs::cost_sorted`](../../../crates/asap-aware-mapping/src/replacement.rs) is where this hooks in today. `search_workload_with` computes each shared -subtree's true `consumer_count` across the whole workload up front (the same +sub-DAG's true `consumer_count` across the whole workload up front (the same role `implement_workload_with`'s pre-pass used to play, before that function was retired along with `bind.rs` — this crate no longer commits to one physically-materialized answer at all; picking and building one final -`SummaryNode` per shared subtree is a downstream deployment's job, not this +`SummaryNode` per shared sub-DAG is a downstream deployment's job, not this crate's). For a `TargetSubDAGCandidates` whose candidates are a -[`SharedSubtreeStrategy`](../../../crates/asap-aware-mapping/src/replacement.rs) +[`SharedSubDAGStrategy`](../../../crates/asap-aware-mapping/src/replacement.rs) share-vs-recompute pair, `cost_sorted`'s ranking step (`rank_group`/ `cse_preference`) asks `CostModel::cse_share_decision` once per group — using one representative bound `SummaryNode` built just for that comparison, not @@ -92,18 +92,18 @@ does not prune them. ## Defaults `cse_recompute_cost`'s default is a structural-size proxy: `cse::dag_node_count`, -the number of *unique* nodes in the subtree's DAG (deduplicated by `Rc` +the number of *unique* nodes in the sub-DAG's DAG (deduplicated by `Rc` pointer identity), not a raw serialization length. This distinction matters -here specifically — a `CseCandidate`'s subtree is, by definition, something -CSE already found sharing in, so it's generally a DAG, not a tree; a naive -tree-shaped size measure (a full `serde_json` serialization, or a recursive -walk with no identity tracking) would re-count any descendant the subtree +here specifically — a `CseCandidate`'s sub-DAG is, by definition, something +CSE already found sharing in, so it generally has internal sharing; a naive +per-path size measure (a full `serde_json` serialization, or a recursive +walk with no identity tracking) would re-count any descendant the sub-DAG already shares internally once per parent that reaches it, over-stating the real cost of holding or recomputing it once. `cse_shared_maintenance_cost`'s default is a small per-`SummaryFamilyType` weight table (exact accumulators cheapest, sketches/samples/wavelets/stat-models progressively more expensive to keep continuously updated) scaled to the same order of magnitude as typical -subtree sizes. Both are documented as coarse heuristic proxies — a real +sub-DAG sizes. Both are documented as coarse heuristic proxies — a real deployment with actual memory/update-cost/query-frequency knowledge overrides either or both, same as `size_params` already lets a deployment override `asap-plan`'s built-in sizing formulas without forking anything else. diff --git a/docs/design_docs/physical-planning-and-deployment.md b/docs/design_docs/physical-planning-and-deployment.md index c13c080af..274e4974c 100644 --- a/docs/design_docs/physical-planning-and-deployment.md +++ b/docs/design_docs/physical-planning-and-deployment.md @@ -30,8 +30,8 @@ associated with the logical DAG, not a separate computation IR. The Logical Post-ASAP DAG is preceded by the Pre-ASAP DAG (`QueryExpr`), the language-independent query semantics before summary selection. Both are -logical. Planning builds Post-ASAP `SummaryNode` trees; `compile_post_asap_dag` -exports the selected tree as a `PostAsapDag`, which is the Physical Plan +logical. Planning builds Post-ASAP `SummaryNode` DAGs; `compile_post_asap_dag` +exports the selected DAG as a `PostAsapDAG`, which is the Physical Plan Compiler's input. Its per-node execution phase (ingestion or query time) is decided by the selected summary maintenance lifecycle, as the layer contract below states. @@ -151,7 +151,7 @@ readiness; those require runtime checks. Physical location, encoding, scheduling and retention are separate execution/deployment contracts. Persisted semantic identity, its wire format and any tenant or dataset binding -belong to the deployment. Planner provides the typed `PostAsapDag` that a +belong to the deployment. Planner provides the typed `PostAsapDAG` that a deployment canonicalizes; it does not define a stored-definition format. ### Running example @@ -319,7 +319,7 @@ The **Physical Plan Compiler** consumes both computation semantics and maintenan requirements: ```text -Logical Post-ASAP DAG (PostAsapDag) +Logical Post-ASAP DAG (PostAsapDAG) + Summary Maintenance Lifecycle + physical capabilities ↓ @@ -433,7 +433,7 @@ frontiers and cost evidence, including updates, retention, recurrence and sharin The lifecycle layer decides timing; physical compilation reads it. Lowering a node does not depend on the frontier, so each query DAG is lowered once and different lifecycle assignments are different cuts of that lowering. -`compile(dag, inputs, roots)` yields the complete `CompiledPhysicalDag`. +`compile(dag, inputs, roots)` yields the complete `CompiledPhysicalDAG`. `frontier_from_timing(&timed_dag)` reads an assignment's timed DAG (from `execution_timed_dag`) and returns its frontier: ingestion-time nodes read by query-time nodes, or an ingestion-time root; a query-time node feeding an diff --git a/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md b/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md index 04c7015f5..520501731 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md +++ b/docs/design_docs/proposals/asap-aware-mapping/analytical-resource-cost.md @@ -58,7 +58,7 @@ The source modules follow those responsibilities rather than treating "analytical" and "streaming" as competing cost systems: ```text -query_physical_lowering.rs ──► EvidenceBackedPhysicalDag +query_physical_lowering.rs ──► EvidenceBackedPhysicalDAG │ physical_operator_statistics.rs ──────┤ physical evidence contract ▼ @@ -75,7 +75,7 @@ analytical_cost.rs ─────────── operator formulas and CPU/m ``` Raw query plans and incrementally maintained summary plans share -`EvidenceBackedPhysicalDag`; there is no streaming-only duplicate of the +`EvidenceBackedPhysicalDAG`; there is no streaming-only duplicate of the physical DAG or operator-statistics contract. Summary-maintenance modules add only the evidence and scheduling semantics that do not exist for an ordinary query plan. @@ -280,7 +280,7 @@ Neither logical IR is the statistics schema: ```text pre-ASAP QueryExpr ─┐ - ├─ physical lowering ─> PhysicalDagNode/PhysicalOperator + ├─ physical lowering ─> PhysicalDAGNode/PhysicalOperator post-ASAP SummaryExpr┘ │ v OperatorStatistics @@ -486,7 +486,7 @@ observed or estimated facts about that operator in this workload. The provider owns provenance, freshness, and derivation. The estimator resolves each reachable node once, so one estimate cannot mix values across a live catalog refresh. A scan has one external source input; every other input is in -the same order as `PhysicalDagNode.children`. Every parent input must equal the +the same order as `PhysicalDAGNode.children`. Every parent input must equal the corresponding child's output in both rows and bytes. Missing node evidence, invalid arity, inconsistent edge dimensions, or a parent/child conflict makes the entire DAG unavailable. `{ rows: 0, bytes: 0 }` is a valid empty logical @@ -495,7 +495,7 @@ positive logical bytes so width-dependent formulas do not invent a row width. A parent therefore cannot silently substitute the original source cardinality for an intermediate edge. -The statistics inputs and `PhysicalDagNode.children` therefore have +The statistics inputs and `PhysicalDAGNode.children` therefore have different arity only for a source leaf: | Operator shape | Statistics inputs | DAG children | @@ -543,7 +543,7 @@ For a selected DAG: 7. Retained summaries remain live across reads. Streaming buffers may be released after their last consumer. -`estimate_physical_dag` implements these rules for `PhysicalDagNode` values, +`estimate_physical_dag` implements these rules for `PhysicalDAGNode` values, one `ComparisonScope`, and an `OperatorStatisticsProvider`. It is a single-plan diagnostic API. Code that ranks a raw and candidate plan must use `estimate_physical_dag_comparison`, which validates exact scope equality before @@ -579,7 +579,7 @@ The parent/child compatibility rules are: | `PerEvaluation` | `PerEvaluation` | valid | | `PerEvaluation` | `Once` | valid only when the child exposes retained state | -For a tree-shaped pipeline, peak memory is normally the maximum live pipeline +For a linear pipeline, peak memory is normally the maximum live pipeline state, not the sum of every node's memory. At a fan-out, join, merge, or nested summary boundary, multiple child states may coexist and must be combined. @@ -591,7 +591,7 @@ deduplicated by physical identity. ### Query-DAG lowering and statistics contract `lower_query_physical_dag` recursively lowers a resolved `Rc` and -returns an `EvidenceBackedPhysicalDag` containing both its nodes and root ID. +returns an `EvidenceBackedPhysicalDAG` containing both its nodes and root ID. It consumes the existing query and physical-operator enums; it does not introduce a parallel logical operator vocabulary. For every occurrence, the lowerer sends a `PhysicalNodeRequest` containing the logical node, selected existing @@ -601,7 +601,7 @@ child physical IDs, and any source coverage to a `physical_id`, the authoritative `OperatorStatistics`, and explicit `output_buffer_bytes`; logical edge bytes are never substituted for an allocation. Missing evidence makes the entire query unavailable. The returned -`EvidenceBackedPhysicalDag` snapshots this evidence so costing does not re-read a live +`EvidenceBackedPhysicalDAG` snapshots this evidence so costing does not re-read a live catalog after lowering. Each lowered Scan is bound to exactly one `SourceCoverage` in the comparison @@ -961,7 +961,7 @@ actual raw target, and then lowers a logical rewrite or requests the fully bound physical DAG for a `SummaryExpr` candidate. The deployment implements `PlannerPhysicalPlanProvider`: query-node evidence is consumed atomically by the generic query lowerer, while summary binding returns a complete -`EvidenceBackedPhysicalDag`, including embedded raw work, build/read operators, retained +`EvidenceBackedPhysicalDAG`, including embedded raw work, build/read operators, retained state, execution multiplicity, and source coverage. The adapter calls `estimate_physical_dag_comparison`; it never calls `DefaultCostModel` or a structural-node-count fallback for final cost. diff --git a/docs/design_docs/proposals/asap-aware-mapping/end-to-end-accuracy-guarantees.md b/docs/design_docs/proposals/asap-aware-mapping/end-to-end-accuracy-guarantees.md index 3a391a53b..d705f6b7a 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/end-to-end-accuracy-guarantees.md +++ b/docs/design_docs/proposals/asap-aware-mapping/end-to-end-accuracy-guarantees.md @@ -489,7 +489,7 @@ a candidate was rejected. provenance, allocations, and rejection reasons. The observable proxy is that a rejected candidate can be diagnosed from exported data without replaying cost ranking. -- **Performance and scalability:** expressions are small trees evaluated during +- **Performance and scalability:** expressions are small DAGs evaluated during candidate construction. No numerical performance claim is made; candidate count and planning latency should be measured before adding richer allocation enumeration. diff --git a/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md b/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md index 065559650..005670db8 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md +++ b/docs/design_docs/proposals/asap-aware-mapping/maintained-populations.md @@ -27,7 +27,7 @@ value is replaced. This distinction requires an explicit rule and state contract an append-only quantile sketch cannot by itself implement current-series updates. The current rule retains an exact population. It does not prescribe a particular -tree, heap or sketch implementation, and it does not imply a deletable DDSketch. +DAG, heap or sketch implementation, and it does not imply a deletable DDSketch. ## Membership semantics diff --git a/docs/design_docs/proposals/asap-aware-mapping/optimizations.md b/docs/design_docs/proposals/asap-aware-mapping/optimizations.md index c5b6dd1a9..ace3838af 100644 --- a/docs/design_docs/proposals/asap-aware-mapping/optimizations.md +++ b/docs/design_docs/proposals/asap-aware-mapping/optimizations.md @@ -232,7 +232,7 @@ The two operators are not identical, but their grouping keys are related. If the aggregation is mergeable, Query B may be derived by rolling up Query A. -This creates reuse opportunities across **hierarchically related groupings**, not just identical subtrees. +This creates reuse opportunities across **hierarchically related groupings**, not just identical sub-DAGs. The legality and cost of this transformation depend on: diff --git a/docs/design_docs/proposals/asapquery-rule-coverage.md b/docs/design_docs/proposals/asapquery-rule-coverage.md index dd6b9a924..e496c9fe9 100644 --- a/docs/design_docs/proposals/asapquery-rule-coverage.md +++ b/docs/design_docs/proposals/asapquery-rule-coverage.md @@ -23,7 +23,7 @@ cost, and selection rules under `optimizer/`. The reviewed source is | Collapsible temporal + spatial aggregates | Semantic-equivalent rewriting | The existing rewrite strategy uses accumulator algebra: sum∘sum, sum∘count, min∘min, and max∘max. It rejects all other pairs and requires identical output schemas. | | Sketch alternatives and exact fallback | Covered more generally | `SketchAlgorithmStrategy` enumerates legal summary realizations. The enclosing memo group always retains the original raw expression as the exact fallback; the strategy does not falsely label an approximate sketch as exact. | | Subpopulation label placement | Covered more generally | `HydraGroupingStrategy` and `GroupingStrategy` express per-subpopulation and shared multi-subpopulation realizations. | -| Shared computation | Covered more generally | workload-wide CSE and `SharedSubtreeStrategy` operate on physical DAG identity rather than AQE names. | +| Shared computation | Covered more generally | workload-wide CSE and `SharedSubDAGStrategy` operate on physical DAG identity rather than AQE names. | | Average decomposition | Semantic-equivalent rewriting | The same rewrite strategy exposes independently optimizable sum/count accumulators when null semantics and schema permit it. | | Merge/delete legality | Covered | Summary-family capabilities and lifecycle validation determine which maintenance operations are legal. | | Window-framework selection | Separate physical-planning work | Window selection must compare an extensible set of implementations, including tumbling, sliding, PromSketch-style exponential-histogram windows, and other window frameworks. This audit does not introduce a closed window enum or choose among them. | @@ -41,7 +41,7 @@ does not create a new strategy category. | Which summary algorithm can implement one aggregate intent | `SketchAlgorithmStrategy` | | How grouping/subpopulation state is laid out | `HydraGroupingStrategy` | | Whether an equivalent logical expression exposes better accumulators | `SemanticEquivalentRewriteStrategy` (the broadened existing avg rewrite; `AvgToSumOverCountStrategy` remains a compatibility name) | -| Whether identical physical work is shared | `SharedSubtreeStrategy` | +| Whether identical physical work is shared | `SharedSubDAGStrategy` | | Whether a finer grouping can answer a coarser grouping | `RollupStrategy` | | Whether tighter accuracy can answer a looser request | `AccuracyReconciliationStrategy` | | Whether a larger Top-K result can answer a smaller limit | `TopKLimitReuseStrategy` | diff --git a/docs/design_docs/proposals/decoupling_op_and_expr.md b/docs/design_docs/proposals/decoupling_op_and_expr.md index 58859db41..fe458bb80 100644 --- a/docs/design_docs/proposals/decoupling_op_and_expr.md +++ b/docs/design_docs/proposals/decoupling_op_and_expr.md @@ -55,7 +55,7 @@ where needed, rather than copying every current `QueryExpr` variant unchanged: Use the canonical [`NonASAPOp` and `OperatorNode` definitions](operator-sharing.md#11-unified-operator-type) from the sharing proposal. Both documents describe the same resolved model: operator inputs and scalar query-result references use `Rc`. -`NonASAPOp` is the payload of an ordinary operator, not a second graph-node type. +`NonASAPOp` is the payload of an ordinary operator, not a second DAG-node type. `BinaryOp` likewise uses the single `BinaryOperator` payload specified there. Names are resolved to `ColumnId` before constructing these nodes. Parsing and @@ -193,7 +193,7 @@ For an open label schema, lowering must retain the complete series identity, including unreferenced labels. If the input provides neither a complete label schema nor a full identity value, this lowering is not valid. `vector(s)` remains a real conversion to a one-element, label-free vector. -Scalar expression trees are owned, while their operator references preserve graph +Scalar expression DAGs are owned, while their operator references preserve DAG identity. The companion's [complete DAG example](operator-sharing.md#13-example-composing-a-logical-dag) shows these expressions inside ordinary operators before and after an ASAP rewrite. @@ -241,7 +241,7 @@ older repository dependency. This documentation change upgrades neither dependen | `DISTINCT`, `UNION`, `INTERSECT`, `EXCEPT`, their supported `ALL` forms | `Dedup`, `Concat` / `SetOp` | **Direct.** Preserve bag multiplicity and SQL duplicate/NULL equality rules. | | `ORDER BY`, `LIMIT`, offset-only queries | `Sort`, `Limit` | **Direct.** Preserve direction and NULL placement; `n = None` means no fetch limit. | | Uncorrelated scalar subquery, `EXISTS`, `IN` / `NOT IN (SELECT ...)` | Explicit scalar plan-reading variants | **Direct.** Preserve the cardinality and NULL contracts in §2.2, including when used in a SELECT list. | -| Derived tables and nonrecursive CTEs | Existing operator subgraphs; aliases resolved to output columns | **Lowered.** Naming alone needs no computation node. Reuse must not alter volatile evaluation. | +| Derived tables and nonrecursive CTEs | Existing operator sub-DAGs; aliases resolved to output columns | **Lowered.** Naming alone needs no computation node. Reuse must not alter volatile evaluation. | ### 3.3 PromQL semantic mapping @@ -318,13 +318,13 @@ of complete SQL/PromQL support. Acceptance requires: - The scalar queries and mixed scalar/vector examples in §2.3 need no `PromqlScalarBridge` or equivalent constant-wrapper node. - Operator dependencies inside scalar conversions/subqueries remain visible and - shared; scalar trees remain owned. Invalid result-kind combinations are rejected. + shared; scalar DAGs remain owned. Invalid result-kind combinations are rejected. - SQL NULL/cardinality rules, PromQL labels and evaluation times survive conversion. Implementation will require frontend, validation and plan-format migration for these explicit structural changes. DDL/DML, session commands, physical execution, new optimization algorithms, accuracy and execution-timing policy are outside this -proposal. The companion document defines the common pre-/post-ASAP operator graph. +proposal. The companion document defines the common pre-/post-ASAP operator DAG. [df-release]: https://github.com/apache/datafusion/releases/tag/55.1.0 [prom-release]: https://github.com/prometheus/prometheus/releases/tag/v3.15.0 diff --git a/docs/design_docs/proposals/operator-sharing.md b/docs/design_docs/proposals/operator-sharing.md index 30bbe3f79..884226238 100644 --- a/docs/design_docs/proposals/operator-sharing.md +++ b/docs/design_docs/proposals/operator-sharing.md @@ -7,7 +7,7 @@ ## Goal and problem Use one operator model before and after ASAP optimization, so ordinary query -operations and summary operations can form one visible computation graph. +operations and summary operations can form one visible computation DAG. Today, the post-ASAP representation wraps relational subplans and duplicates some relational operators outside those wrappers. This causes three problems: @@ -19,7 +19,7 @@ relational operators outside those wrappers. This causes three problems: children. For example, consider a p99 latency query that projects its input columns, builds a -KLL summary, and projects the estimated result. The trees below read from the result +KLL summary, and projects the estimated result. The DAGs below read from the result at the top to the data source at the bottom: ```text @@ -33,7 +33,7 @@ Post-ASAP projection Project ``` Today the two projections need separate representations, and the scan is hidden -inside the wrapped subplan. In the proposed graph, both projections use the same +inside the wrapped subplan. In the proposed DAG, both projections use the same operator definition and the scan is directly visible. A union can likewise consume summary estimates without needing a separate post-ASAP union definition. @@ -56,7 +56,7 @@ operation variants and schema internals are expanded afterward. These declaratio are shared by the detailed sections, not separate abbreviated types. ```rust -// A graph node combines its operation with common planning properties (§2). +// A DAG node combines its operation with common planning properties (§2). struct OperatorNode { operator: Operator, result_kind: OperatorResultKind, @@ -72,14 +72,14 @@ enum Operator { } // NonASAPOp / ASAPOp: detailed below; their inputs are Rc. -// ScalarExpr: an owned value-expression tree, defined in the companion proposal. +// ScalarExpr: an owned, unshared value expression, defined in the companion proposal. // Schema / OperatorResultKind: defined in §2.1. ``` An operator owns its scalar expressions and references input nodes through `Rc`. Either operation category can consume the other's outputs when the input contract permits it. `NonASAP` describes one operation, not its entire -subgraph. Frontend graphs contain only NonASAP operations; ASAP optimization may +sub-DAG. Frontend DAGs contain only NonASAP operations; ASAP optimization may introduce state construction and readout. | Category | Meaning | All operations | @@ -199,13 +199,13 @@ A filter is an operator because it transforms a table. Its predicate, such as Scalar expressions belong to an operator field or a scalar query. Predicates, projection expressions and sort keys describe value computation in that context. Explicit scalar conversions and subqueries may reference operators; those are -visible graph dependencies with defined cardinality rules. This prevents an +visible DAG dependencies with defined cardinality rules. This prevents an arbitrary expression from being mistaken for a table-producing plan. The [companion proposal](decoupling_op_and_expr.md) defines this distinction. The companion's `ScalarExpr` uses `Rc` for `PromqlScalarFromVector`, `ScalarSubquery`, `Exists` and `InSubquery`, so those expressions already reference -this common graph before and after optimization. +this common DAG before and after optimization. In `scalar(sum(up))`, `scalar()` is Prometheus PromQL's built-in vector-to-scalar function, explicitly written by the query author. This proposal does not insert @@ -214,9 +214,9 @@ The scalar expression `PromqlScalarFromVector` represents that function and refe result to obtain one number. A valid ASAP rewrite may replace that producer with a summary readout, preserving the required vector and accuracy semantics; it cannot substitute raw summary state. Ordinary expressions such as `price * 2` reference -columns and literals, not a query subgraph. +columns and literals, not a query sub-DAG. -These are **query subgraphs referenced by scalar expressions**, with the same +These are **query sub-DAGs referenced by scalar expressions**, with the same producer identity as any other operator dependency. ### 1.3 Example: composing a logical DAG @@ -292,7 +292,7 @@ existing capability and rewrite checks permit that exact implementation. | Part of the design | Role in this example | |---|---| -| `OperatorNode` | Every graph node, holding its operation and common result/schema, guarantee and timing properties. | +| `OperatorNode` | Every DAG node, holding its operation and common result/schema, guarantee and timing properties. | | `Operator` | Selects the `NonASAP` or `ASAP` operation category in each node. | | `NonASAPOp` | Scan, filter, aggregate and projection before optimization; scan, filter and projection still use these definitions afterward. | | `ASAPOp` | Builds accumulator state and finalizes it after the rewrite. | @@ -306,7 +306,7 @@ For this example, assume `bytes` is nullable `Int64`. The output metadata is: | Aggregate before optimization | `Relation` | `sum_bytes: Plain(Int64)`, nullable | | Summary build after optimization | `State` | `sum_state: ExactAggregate(Sum, Sum)`, non-null accumulator state | | Finalize after optimization | `Relation` | `sum_bytes: Plain(Int64)`, nullable | -| Project in either graph | `Relation` | `total_bytes: Plain(Int64)`, nullable | +| Project in either DAG | `Relation` | `total_bytes: Plain(Int64)`, nullable | The empty accumulator finalizes to SQL NULL; the accumulator itself is state, not a nullable numeric value. The projection consumes the finalized column. Guarantees @@ -315,7 +315,7 @@ physical planning. The topmost Project node produces the query result. This illustrates the connection between the two proposals: scalar separation makes predicates and value expressions explicit; operator unification lets those -same ordinary operations consume ASAP results through normal graph edges. +same ordinary operations consume ASAP results through normal DAG edges. ### 1.4 Scope of operator sharing @@ -452,11 +452,11 @@ contract of `PromqlScalarFromVector` and other scalar plan reads. | Validation entry | Scope and stage | |---|---| -| `OperatorNode::validate_structure()` | Walks the reachable operator graph, including scalar plan references; checks input contracts, scalar typing and agreement between retained and derived output metadata. Valid for logical and physical plans; permits `timing = None`. | +| `OperatorNode::validate_structure()` | Walks the reachable operator DAG, including scalar plan references; checks input contracts, scalar typing and agreement between retained and derived output metadata. Valid for logical and physical plans; permits `timing = None`. | | `OperatorNode::validate_execution_timing()` | Includes structural validation, then requires assigned timing on every executable operator and checks phase dependencies. Used for executable physical candidates. | | Existing planner assessment and selection (#509) | Establishes guarantees using the existing accuracy models and checks them against request requirements and deployment capabilities. Neither node method re-proves a guarantee or decides request feasibility. | -The two node methods need only the graph and its annotations. Request requirements +The two node methods need only the DAG and its annotations. Request requirements and deployment models remain inputs to the existing planning/selection workflow, not implicit globals of `validate_structure`. Passing the timing check alone does not establish that a physical candidate satisfies the query's accuracy requirement. @@ -521,7 +521,7 @@ The representation must preserve the resulting execution constraints: ingestion- work cannot depend on query-time results, and consumers must receive values or state that are available when needed. Materialization choices, retention and plan selection remain governed by #509; this document does not define another lifecycle policy. -These constraints also apply to query subgraphs referenced by scalar expressions. +These constraints also apply to query sub-DAGs referenced by scalar expressions. PromQL evaluation timestamps and SQL statement time are separate from these execution phases. `TimeShift`, subquery grids and `EvalTimestamp` retain their @@ -534,11 +534,11 @@ The design is successful when: - A projection uses the same semantics above and below summary computations. - A union or another ordinary operator can consume summary estimates on its inputs. -- Unifying the representation preserves existing graph dependencies, including +- Unifying the representation preserves existing DAG dependencies, including any shared inputs; it does not introduce new sharing rules. - Existing value/state, accuracy and execution constraints remain enforceable on the unified representation. - Scalar expressions and conversions use the same representation before and after optimization, with no bridge nodes or hidden subplans. -- Structural and timing validation include query subgraphs referenced by scalar +- Structural and timing validation include query sub-DAGs referenced by scalar expressions; planner assessment includes their accuracy dependencies. diff --git a/docs/design_docs/proposals/univmon-frequency-summary.md b/docs/design_docs/proposals/univmon-frequency-summary.md index 21a4cb3d8..75cfd080b 100644 --- a/docs/design_docs/proposals/univmon-frequency-summary.md +++ b/docs/design_docs/proposals/univmon-frequency-summary.md @@ -30,11 +30,11 @@ The parameter contract records heap size, sketch rows, sketch columns and number of layers. Default dimensions define a candidate configuration, not an epsilon guarantee. A deployment accuracy model must supply calibrated evidence before an approximate readout can satisfy an accuracy target. Without -that evidence, the Planner keeps the exact subtree. HLL/Theta/KMV remain +that evidence, the Planner keeps the exact sub-DAG. HLL/Theta/KMV remain cardinality alternatives, and exact count remains the cheaper first count candidate. -All four readouts have the same unit-weight update, input subtree, grouping, +All four readouts have the same unit-weight update, input sub-DAG, grouping, window, parameter identity and state schema. Existing post-ASAP structural sharing can therefore intern their state producer while preserving distinct readout nodes. Sharing is only legal within the same execution/data scope. diff --git a/docs/develop_docs/asap-aware-mapping-architecture.md b/docs/develop_docs/asap-aware-mapping-architecture.md index f7e2a9a7b..5170f7d91 100644 --- a/docs/develop_docs/asap-aware-mapping-architecture.md +++ b/docs/develop_docs/asap-aware-mapping-architecture.md @@ -59,9 +59,9 @@ Terminology used in the diagram: those queries. **Pre-ASAP** means this logical input form, before the planner realizes an operation as a concrete ASAP realization; **post-ASAP** means the resulting realization form. -- A **DAG** (directed acyclic graph) represents query operators whose subtrees +- A **DAG** (directed acyclic graph) represents query operators whose sub-DAGs may be shared. **CSE** (common subexpression elimination) finds equivalent - subtrees and represents legal reuse by making them the same shared node. + sub-DAGs and represents legal reuse by making them the same shared node. Rust's `Rc` (reference-counted pointer) records that shared node identity. - A **target** is one replaceable site. A **candidate** is one valid alternative for it. `Replacement::Summary` is a constructed post-ASAP summary—maintained state @@ -120,7 +120,7 @@ flowchart TB ``` The generic `ReplacementStrategy` box is the extension point. The default -registry supplies summary realization, Hydra grouping, shared-subtree, +registry supplies summary realization, Hydra grouping, shared-sub-DAG, average-rewrite and exact-composition strategies. Section 3.3 describes the registries and the workload-derived roll-up rule. @@ -138,7 +138,7 @@ complete alternative set has been built. Use `search_workload` or `search_workload_with` for normal planner search. The search performs these steps: -1. Run CSE once to merge structurally identical subtrees that may legally be +1. Run CSE once to merge structurally identical sub-DAGs that may legally be shared. 2. Walk the complete DAG beneath every query root, including nodes below unshared parents. @@ -158,9 +158,9 @@ flowchart LR classDef common fill:#fff6dd,stroke:#b78922,color:#513d0c ROOTS["Input
one or more named QueryExpr roots"]:::workload - ROOTS --> CSE["Canonicalize sharing
merge structurally identical, legally shareable subtrees"]:::workload + ROOTS --> CSE["Canonicalize sharing
merge structurally identical, legally shareable sub-DAGs"]:::workload CSE --> WALK["Discover sites
walk the complete DAG, including nodes below unshared parents"]:::workload - WALK --> T["Build TargetSubDAG
retain the subtree's Rc identity and measured consumer_count"]:::workload + WALK --> T["Build TargetSubDAG
retain the sub-DAG's Rc identity and measured consumer_count"]:::workload T --> MATCH MATCH["matches(target)
cheaply decide whether this strategy has alternatives"]:::common MATCH -->|"true"| REPLACE["propose(target)
construct supported legal alternatives;
retain structured accuracy rejections"]:::common @@ -169,7 +169,7 @@ flowchart LR ``` `consumer_count` is workload information, not an estimate of runtime -executions. It matters to strategies such as `SharedSubtreeStrategy`, which +executions. It matters to strategies such as `SharedSubDAGStrategy`, which only has a share-versus-recompute choice when a target has multiple consumers. ### 3.2 Generate candidates through `ReplacementStrategy` @@ -205,7 +205,7 @@ The default context-free registry contains five `ReplacementStrategy` implementa realizations. Candidates are sized and ordered for the target's accuracy requirement; candidates without a sufficient guarantee are rejected before costing. -- `SharedSubtreeStrategy` uses `consumer_count` to identify shared targets. It +- `SharedSubDAGStrategy` uses `consumer_count` to identify shared targets. It emits both build-once-and-share and recompute-independently rewrites when a target has multiple consumers. - `HydraGroupingStrategy` proposes eligible shared multi-subpopulation layouts. diff --git a/docs/develop_docs/asap-aware-mapping-contracts.md b/docs/develop_docs/asap-aware-mapping-contracts.md index d1245366f..d642cbfb5 100644 --- a/docs/develop_docs/asap-aware-mapping-contracts.md +++ b/docs/develop_docs/asap-aware-mapping-contracts.md @@ -28,7 +28,7 @@ For example, consider two top-level queries: - `sum by (service) (rate(m[5m]))` - `avg by (service) (rate(m[5m]))` -After `share_common_subtrees` merges their identical `rate(m[5m])` subtrees, both query trees point to the same `Rc`. That node's `consumer_count` is `2`, regardless of how often either query executes. +After `share_common_sub_dags` merges their identical `rate(m[5m])` sub-DAGs, both query DAGs point to the same `Rc`. That node's `consumer_count` is `2`, regardless of how often either query executes. Use: @@ -85,7 +85,7 @@ is a `Summary`; KLL (Karnin–Lang–Liberty) is a quantile-sketch algorithm. ```text compute independently vs. -reuse an already shared logical subtree +reuse an already shared logical sub-DAG ``` is represented as a `Rewrite`. @@ -253,7 +253,7 @@ bounds, but does not execute workloads or own deployment measurements. Most hook fn readout_extension(&self, ext_kind: &str, payload: &serde_json::Value, col: &ColumnRef) -> SketchQuery; ``` -- **`cse_recompute_cost`** — estimate the one-time cost of recomputing a CSE candidate's subtree independently at a single consumer. Default: `default_cse_recompute_cost`, a structural-size proxy. +- **`cse_recompute_cost`** — estimate the one-time cost of recomputing a CSE candidate's sub-DAG independently at a single consumer. Default: `default_cse_recompute_cost`, a structural-size proxy. ```rust fn cse_recompute_cost(&self, candidate: &CseCandidate) -> Cost; @@ -311,9 +311,9 @@ pub struct RankedTargetSubDAGCandidates<'a> { } ``` -`search_workload(roots)` runs the shared-subtree pass once, discovers every target across every root's whole DAG (not just root-level sharing — a `SharedSubtreeStrategy` candidate three levels under an unshared `Filter` is exactly as real a site as a shared whole root), and asks every registered strategy to a fixpoint. Two logically different candidates at two different targets are never copied into two separate plans — they're two entries in two different `TargetSubDAGCandidates`s, sharing every other node in the workload by construction. +`search_workload(roots)` runs the shared-sub-DAG pass once, discovers every target across every root's whole DAG (not just root-level sharing — a `SharedSubDAGStrategy` candidate three levels under an unshared `Filter` is exactly as real a site as a shared whole root), and asks every registered strategy to a fixpoint. Two logically different candidates at two different targets are never copied into two separate plans — they're two entries in two different `TargetSubDAGCandidates`s, sharing every other node in the workload by construction. -`CandidateLogicalASAPDAGs::cost_sorted(cost_model)` is the one ranking step: for each candidate set, it dispatches by candidate shape — a same-shape `Rewrite` pair (a `SharedSubtreeStrategy` share/recompute choice) goes through `CostModel::cse_share_decision`; a same-shape run of `Summary` candidates realizing sketches (a `SketchAlgorithmStrategy` choice) goes through `CostModel::rank_candidates`; and a mixed candidate set is ordered by each candidate's `CostModel::estimate_cost`. Every candidate gets a numeric cost aligned index-for-index in `costs`. Count in, count out—nothing is dropped to produce a ranking. Legality checks +`CandidateLogicalASAPDAGs::cost_sorted(cost_model)` is the one ranking step: for each candidate set, it dispatches by candidate shape — a same-shape `Rewrite` pair (a `SharedSubDAGStrategy` share/recompute choice) goes through `CostModel::cse_share_decision`; a same-shape run of `Summary` candidates realizing sketches (a `SketchAlgorithmStrategy` choice) goes through `CostModel::rank_candidates`; and a mixed candidate set is ordered by each candidate's `CostModel::estimate_cost`. Every candidate gets a numeric cost aligned index-for-index in `costs`. Count in, count out—nothing is dropped to produce a ranking. Legality checks may already have removed proposals before this boundary. In particular, `search_workload_with_targets` checks explicit per-root targets, while retaining direct DDSketch ratios with missing domain evidence and no root guarantee for @@ -379,14 +379,14 @@ The crate provides no default `Matcher` implementation because the answer depend Concretely, `explanation.rs` reports three candidate kinds from each `TargetSubDAGCandidates`: - `ExplanationKind::SketchApproximation` — the set contains a `Replacement::Summary` that realizes `SummaryFamilyType::Sketch(..)`, not just an exact/pass-through candidate. -- `ExplanationKind::CommonSubexpressionReuse` — `consumer_count >= 2` and the set contains `SharedSubtreeStrategy`'s "build once and share" candidate (the `Replacement::Rewrite` whose `Rc` is the set's `target`). +- `ExplanationKind::CommonSubexpressionReuse` — `consumer_count >= 2` and the set contains `SharedSubDAGStrategy`'s "build once and share" candidate (the `Replacement::Rewrite` whose `Rc` is the set's `target`). - `ExplanationKind::ExactComposition` — the candidate set contains an exact operation composed with a child target whose realization remains a coordinated choice. Each `ReplacementExplanation::reason` is copied verbatim from the matching candidate's own `ReplacementSubDAG::rationale`. Nothing in `explanation.rs` re-explains why a candidate is valid; that explanation already exists exactly once, on the candidate itself. -`ReplacementExplanation` carries both `node_hash` and `target`. A downstream consumer first compares `node_hash` with an exported `DagNode::hash` to narrow the search, then compares the exact target expression with the node's in-process source expression. This preserves the hash's role as a fast filter while making the final association collision-safe; `location` remains human-readable presentation text rather than a machine identifier. +`ReplacementExplanation` carries both `node_hash` and `target`. A downstream consumer first compares `node_hash` with an exported `DAGNode::hash` to narrow the search, then compares the exact target expression with the node's in-process source expression. This preserves the hash's role as a fast filter while making the final association collision-safe; `location` remains human-readable presentation text rather than a machine identifier. ### Why there is no `ExplanationRule` trait diff --git a/docs/develop_docs/extend-asap-aware-mapping.md b/docs/develop_docs/extend-asap-aware-mapping.md index 235da5460..c8c1b4973 100644 --- a/docs/develop_docs/extend-asap-aware-mapping.md +++ b/docs/develop_docs/extend-asap-aware-mapping.md @@ -65,7 +65,7 @@ fn matches(&self, target: &TargetSubDAG<'_>) -> bool { } ``` -is enough for the current shared-subtree strategy. +is enough for the current shared-sub-DAG strategy. #### Guideline @@ -293,9 +293,9 @@ aggregate choices remain independent. --- -### Example: current `SharedSubtreeStrategy` +### Example: current `SharedSubDAGStrategy` -`SharedSubtreeStrategy` is the reference implementation for a logical rewrite strategy. +`SharedSubDAGStrategy` is the reference implementation for a logical rewrite strategy. It applies when: @@ -331,9 +331,9 @@ That preference belongs to the cost model. share-versus-recompute candidate pair. The strategy still returns both alternatives because enumeration and ranking are separate steps: -- `consumer_count >= 2` means `share_common_subtrees` has already merged the expression into one shared `Rc`. The shared alternative is therefore an `Rc::clone`; the independent alternative requires a deep clone. +- `consumer_count >= 2` means `share_common_sub_dags` has already merged the expression into one shared `Rc`. The shared alternative is therefore an `Rc::clone`; the independent alternative requires a deep clone. - `cse_share_decision` is used by the ranking path, not by - `SharedSubtreeStrategy`. + `SharedSubDAGStrategy`. - The strategy must return both valid alternatives even if the current cost model strongly prefers one. A future whole-plan search may choose differently from today's local comparison. This example is useful when implementing transformations such as: @@ -462,7 +462,7 @@ assert!( For a strategy whose explanation includes important context, also test that context. -For example, the shared-subtree tests verify that the consumer count appears in the rationale. +For example, the shared-sub-DAG tests verify that the consumer count appears in the rationale. --- @@ -470,7 +470,7 @@ For example, the shared-subtree tests verify that the consumer count appears in For logical rewrites, test the structural property that distinguishes the alternatives. -For example, the current shared-subtree tests verify: +For example, the current shared-sub-DAG tests verify: ```rust Rc::ptr_eq(shared, &q) @@ -660,7 +660,7 @@ This complements `realize_extension`: realization defines what gets maintained; #### `cse_recompute_cost` -Use to estimate the cost of computing a common subtree independently at each consumer. +Use to estimate the cost of computing a common sub-DAG independently at each consumer. ```rust fn cse_recompute_cost( @@ -673,7 +673,7 @@ fn cse_recompute_cost( #### `cse_shared_maintenance_cost` -Use to estimate the cost of computing and maintaining a shared subtree. +Use to estimate the cost of computing and maintaining a shared sub-DAG. ```rust fn cse_shared_maintenance_cost( diff --git a/docs/develop_docs/library-api.md b/docs/develop_docs/library-api.md index b1d1226b0..719de6ba2 100644 --- a/docs/develop_docs/library-api.md +++ b/docs/develop_docs/library-api.md @@ -271,7 +271,7 @@ before physical selection; do not treat their presence as deployment permission. ```text CandidateLogicalASAPDAGs::enumerate_candidate_dags_for_root(&self, id: &Id, expansion_limit: usize) - -> Result, RealizationError> + -> Result, RealizationError> ``` Returns every distinct finalized DAG for one root, unranked; other roots' @@ -302,7 +302,7 @@ pass. An omitted strategy contributes no proposals of its own. | --- | --- | --- | | `SketchAlgorithmStrategy::new(&model)` | Enumerates supported exact/sketch implementations and parameter choices for aggregate targets | Yes | | `HydraGroupingStrategy::new(&model)` | Considers a shared multi-subpopulation structure for supported grouped sketch families, subject to accuracy evidence | Yes | -| `SharedSubtreeStrategy` | Proposes sharing versus independent recomputation at reused subtrees | Yes | +| `SharedSubDAGStrategy` | Proposes sharing versus independent recomputation at reused sub-DAGs | Yes | | `SemanticEquivalentRewriteStrategy` | Proposes supported equivalent aggregate rewrites, including decomposing average into sum/count | Yes | | `ExactCompositionStrategy::new(&model)` | Proposes supported exact operations around summary readouts or in maintenance | Yes | | Your `ReplacementStrategy` implementation | Adds domain-specific legal replacement proposals | No | @@ -353,7 +353,7 @@ use asap_types::workload::{ }; use asap_aware_mapping::{ search_workload_with_targets, DefaultAccuracyModel, DefaultCostModel, - ReplacementStrategy, SketchAlgorithmStrategy, SharedSubtreeStrategy, + ReplacementStrategy, SketchAlgorithmStrategy, SharedSubDAGStrategy, }; use asap_types::types::AccuracyTarget; @@ -387,7 +387,7 @@ fn main() -> Result<(), Box> { let model = DefaultCostModel; let strategies: Vec> = vec![ Box::new(SketchAlgorithmStrategy::new(&model)), - Box::new(SharedSubtreeStrategy), + Box::new(SharedSubDAGStrategy), ]; let space = search_workload_with_targets( vec![("q1", root, Some(accuracy))], &strategies, &DefaultAccuracyModel, @@ -655,7 +655,7 @@ from that hook, as in Planner selection. A lifecycle choice then fixes each physical placement through timing: a continuously maintained state and its inputs run at ingestion time, while an -ephemeral one stays at query time. Compile each query's `PostAsapDag` once and +ephemeral one stays at query time. Compile each query's `PostAsapDAG` once and cut every chosen assignment from that result: ```rust @@ -701,7 +701,7 @@ in the workload**. Here, “global” describes that cross-target scope. It does mean a proven globally optimal solution over every possible physical plan, nor selection across every machine in a deployment. -Consider this conceptual dependency graph: +Consider this conceptual dependency DAG: ```text Q1 --+ @@ -752,7 +752,7 @@ GlobalSelection::assemble_selected_dag(&self, target: &Rc) ``` For structural inspection only, this complete example selects a semantic root -and exports its inspection graph. It performs no lifecycle or deployment planning. +and exports its inspection DAG. It performs no lifecycle or deployment planning. Use lifecycle-aware selection above when the comparison needs those decisions. ```rust @@ -795,8 +795,8 @@ fn main() -> Result<(), Box> { let selection = space.global_selection(&DefaultCostModel); // Search may canonicalize roots; use the root returned by CandidateLogicalASAPDAGs. if let Some(summary) = selection.assemble_selected_dag(&space.roots[0].1)? { - let graph = asap_types::dag_export::export_summary(&summary); - println!("{graph:#?}"); + let dag = asap_types::dag_export::export_summary(&summary); + println!("{dag:#?}"); } Ok(()) } @@ -806,14 +806,14 @@ fn main() -> Result<(), Box> { | Function/type | Purpose | | --- | --- | -| `asap_types::dag_export::export(&query)` | Pre-ASAP inspection graph | -| `asap_types::dag_export::export_summary(&summary)` | Post-ASAP inspection graph | +| `asap_types::dag_export::export(&query)` | Pre-ASAP inspection DAG | +| `asap_types::dag_export::export_summary(&summary)` | Post-ASAP inspection DAG | | `asap_types::post_asap::compile_post_asap_dag(&root)` | Compile a semantic DAG with execution-data-state validation; not a physical plan | -| `PostAsapDagDocument::new(dag)` and `.validate()` | Versioned semantic envelope and explicit validation; constructing it alone does not validate | -| `asap_aware_mapping::export_summary_maintenance_plan(&plan)` | Graph plus lifecycle deployments, alternatives and available cost/guarantee information | +| `PostAsapDAGDocument::new(dag)` and `.validate()` | Versioned semantic envelope and explicit validation; constructing it alone does not validate | +| `asap_aware_mapping::export_summary_maintenance_plan(&plan)` | DAG plus lifecycle deployments, alternatives and available cost/guarantee information | | `explain_replacements` / `explain_replacements_with` | Findings from default/custom-strategy search; not a complete physical feasibility report | -Choose the export matching your intended handoff: an inspection graph is not +Choose the export matching your intended handoff: an inspection DAG is not interchangeable with a versioned execution contract. Preserve lifecycle and cost/guarantee evidence needed downstream instead of exporting only a bare DAG. For public symbol details, build local API documentation with: diff --git a/docs/develop_docs/metrics-observability-corpora.md b/docs/develop_docs/metrics-observability-corpora.md index 3dd1d2b8f..d127e92ce 100644 --- a/docs/develop_docs/metrics-observability-corpora.md +++ b/docs/develop_docs/metrics-observability-corpora.md @@ -62,7 +62,7 @@ query root. It does not measure workload-wide search or the other default strategies. The default workload search currently registers `SketchAlgorithmStrategy`, -`HydraGroupingStrategy`, `SharedSubtreeStrategy`, and +`HydraGroupingStrategy`, `SharedSubDAGStrategy`, and `AvgToSumOverCountStrategy`. Workload context can additionally contribute `RollupStrategy` and `AccuracyReconciliationStrategy`. This baseline is therefore a sketch-only comparison point. diff --git a/docs/develop_docs/native-promql-inputs.md b/docs/develop_docs/native-promql-inputs.md index aa1a52b63..d263b7825 100644 --- a/docs/develop_docs/native-promql-inputs.md +++ b/docs/develop_docs/native-promql-inputs.md @@ -48,7 +48,7 @@ tests, not proof of Backend candidate selection or durable deployment execution. Spatial heap candidates use the same complete series identity. Planner's `current_series_topk_candidates` explores a CountSketch-with-heap realization -of canonical Sort/Limit under an explicit accuracy target. The physical graph +of canonical Sort/Limit under an explicit accuracy target. The physical DAG selects the latest eligible samples before building a fresh heap. A maintained population boundary can supply that snapshot directly. Arbitrary signed metric values do not authorize CMS; counter Rate's non-negative proof is separate. diff --git a/docs/develop_docs/physical-compile-coverage.md b/docs/develop_docs/physical-compile-coverage.md index 3fa54fb4e..5d048d7f7 100644 --- a/docs/develop_docs/physical-compile-coverage.md +++ b/docs/develop_docs/physical-compile-coverage.md @@ -6,12 +6,12 @@ Audience: developers moving computation from ASAPQuery-backend into ## Contract Logical selection decides what to compute. The maintenance lifecycle sets node -timing. `physical_planner::compile` turns a timed `PostAsapDag` into physical +timing. `physical_planner::compile` turns a timed `PostAsapDAG` into physical operator DAGs. The backend owns ingestion, panes, storage, stored-state readout, external exact engines, pricing/selection, and execution scheduling. A backend lowering is *covered* when `compile` accepts the corresponding -`PostAsapDag` node and produces operators with the same result. The backend +`PostAsapDAG` node and produces operators with the same result. The backend should then pass the timed DAG and its input contracts to `compile`. It should not rebuild operator choices from PromQL text or construct operators itself. @@ -29,7 +29,7 @@ Status values: | # | Backend site | Computation | Planner node | Status at #475 | Notes | |---|---|---|---|---|---| -| 1 | `query_time.rs` `Lower::lower`, `compile_logical` | PromQL AST → `QueryTimeOperator` graph for a native query | `Fallback { QueryExpr }` subtrees plus value payloads | Missing | `compile` lowers `Fallback` only as a raw `Scan` source. | +| 1 | `query_time.rs` `Lower::lower`, `compile_logical` | PromQL AST → `QueryTimeOperator` DAG for a native query | `Fallback { QueryExpr }` sub-DAGs plus value payloads | Missing | `compile` lowers `Fallback` only as a raw `Scan` source. | | 2 | `QueryTimeOperator::Aggregate` (sum/min/max/avg/count) | Grouped value aggregation | `Value::Exact(Aggregate)`; `SummaryAgg{ExactAggregate, Reduce}` over finalized values | Supported | Also `promql_values::compile_aggregate`. | | 3 | `QueryTimeOperator::Sort`, `Limit` (topk, sort, sort_desc) | Ordering and per-group limits | `Value::Sort`, `Value::Limit` | Supported | | | 4 | `QueryTimeOperator::Binary`, `QueryPlanNode::Binary` (vector ⊗ scalar) | Arithmetic with a scalar operand | `Binary` whose operand is `Fallback{PromqlScalarBridge(Literal)}` | Missing | Query-time `Binary` accepts only label-map vector schemas. The literal node has no native binding. | @@ -41,14 +41,14 @@ Status values: | 10 | `QueryTimeOperator::HistogramQuantile` | Bucket interpolation | `Fallback` / `AggIntent::HistogramQuantile` | Missing | Only `promql_values::compile_histogram_quantile`. | | 11 | `QueryTimeOperator::Temporal` (rate, increase, `*_over_time`) | Per-series window functions | `SummaryAgg{PerEntity}` over `TimeRange(Scan)` | Partial | Supported with closed series identity. Not supported over `Fallback` matrices (`compile_temporal` only). | | 12 | `logical_dag.rs` `Subquery`, `subquery_grid`, `expanded_inputs` | Re-evaluate the child on a step grid and assemble a matrix | `Fallback{PromqlSubquery}` | Missing | No Planner operator. | -| 13 | `QueryPlanNode::Scalar`, `DagCompiler::lower` scalar literal | Scalar constant | `Fallback{PromqlScalarBridge(Literal)}` | Missing | Only `promql_values::compile_scalar`. | -| 14 | `DagCompiler::lower` `ReduceSum`; `physical_values.rs` PerEntity projection | Sum over finalized values; per-entity identity | `SummaryAgg{ExactAggregate(Sum)}` | Supported | The backend builds an identity `Operator::project` itself for PerEntity. | -| 15 | `DagCompiler::lower` `ExactReadout`; `post_asap_readout.rs` ExactReadout | Finalize exact state (sum/count/min/max/rate/increase) | `Value::FinalizeExactAccumulator` | Partial | Count yields Int64 against a declared Float64 PromQL value. `compile` rejects it. | +| 13 | `QueryPlanNode::Scalar`, `DAGCompiler::lower` scalar literal | Scalar constant | `Fallback{PromqlScalarBridge(Literal)}` | Missing | Only `promql_values::compile_scalar`. | +| 14 | `DAGCompiler::lower` `ReduceSum`; `physical_values.rs` PerEntity projection | Sum over finalized values; per-entity identity | `SummaryAgg{ExactAggregate(Sum)}` | Supported | The backend builds an identity `Operator::project` itself for PerEntity. | +| 15 | `DAGCompiler::lower` `ExactReadout`; `post_asap_readout.rs` ExactReadout | Finalize exact state (sum/count/min/max/rate/increase) | `Value::FinalizeExactAccumulator` | Partial | Count yields Int64 against a declared Float64 PromQL value. `compile` rejects it. | | 16 | `post_asap_readout.rs` SummaryEstimate (`readout_bound`, `expand_item_rows`) | Sketch estimate per group; TopK item expansion | `SummaryEstimate` | Partial | The backend's label-map state layout and MetricsQL `__name__` rules have no Planner equivalent. `compile_exact_readout` has no sketch counterpart. | | 17 | `post_asap_readout.rs` SummaryMerge (`merge_bound_states`) | Merge states by group | `SummaryMerge` | Supported | Union plus `summary_merge`. | | 18 | `post_asap_readout.rs` counter range parameters | Counter lookback for rate/increase | `TimeRange` ancestor of finalization | Supported | Applied through `with_counter_lookback`. | | 19 | `post_asap_readout.rs` `execute_value_fragment` | Per-timestamp binding of a value fragment | n/a | Backend | Evaluation scheduling. | -| 20 | `DagCompiler::lower` SummaryJoin / Subtract / Delete | Summary algebra | `SummaryJoin`, `SummarySubtract`, `SummaryDelete` | Missing | The backend also rejects these (`ExactFallback`). | +| 20 | `DAGCompiler::lower` SummaryJoin / Subtract / Delete | Summary algebra | `SummaryJoin`, `SummarySubtract`, `SummaryDelete` | Missing | The backend also rejects these (`ExactFallback`). | | 21 | `current_series.rs` Snapshot + TopK | Current-series ranking | `ReadPopulation{TopK}` | Supported | | | 22 | `current_series.rs` Sum / Count / Average | Current-series aggregates | `ReadPopulation{Sum,Count,Average}` | Missing | `compile` accepts only TopK. | | 23 | `current_series.rs` Quantile | Current-series quantile | `ReadPopulation{Quantile}` | Missing | No exact quantile reduction. | @@ -56,7 +56,7 @@ Status values: | 25 | `raw_dag.rs` weight `Constant` | Unit/constant-weight update | `SummaryAgg` | Missing | `compile_node` requires a column weight. | | 26 | `raw_dag.rs` item `Column` / `Tuple` | Keyed update item | `SummaryAgg{item}` | Supported | `keyed_summary_build`. | | 27 | `raw_dag.rs` item `EntityIdentity` | Series-identity item | `SummaryAgg{item}` | Missing | Needs the series-identity column. | -| 28 | `physical_values.rs` `compile`, `combine` | Translate `QueryTimeOperator` to `promql_values::*`; compose fragments | n/a | Supported | Exists only because of row 1. `CompiledPhysicalDag::compose` is Planner API. | +| 28 | `physical_values.rs` `compile`, `combine` | Translate `QueryTimeOperator` to `promql_values::*`; compose fragments | n/a | Supported | Exists only because of row 1. `CompiledPhysicalDAG::compose` is Planner API. | | 29 | `query_plan.rs` `compile_native_fragment` (Semi join, Exact aggregate, Sort, Limit, Filter) | Relational value ops | `RelationalJoin`, `Value::*` | Supported | Already calls `compile`. | | 30 | `query_time.rs` `selected_query_time_nodes`, `selected_native_expression`, `selected_aggregate_operator` | Recover operator identity from original PromQL text | Payload variants (`ExactKind::Min`/`Max`, `AggIntent`) | Supported | Payloads already carry the identity. These witnesses are needed only while row 1 remains. | | 31 | Scan, ExactSubquery, CandidateExactSubquery, CurrentSeries ingest, ReadMaterialization, ExternalExact | Storage reads and external engines | Input contracts | Backend | | @@ -188,7 +188,7 @@ Totals after this change: 20 Supported, 5 Partial, 4 Missing, 2 Backend. An argument whose output provably lacks `le`, such as `sum by (job) (rate(x_bucket[5m]))`, is rejected at lowering. Prometheus returns an empty vector for it. Candidate search keeps the classic form as one -exact `KeepPreAsap` subtree for every accuracy target; it has no sketch +exact `KeepPreAsap` sub-DAG for every accuracy target; it has no sketch candidate. `histogram_quantiles` lowers each branch the same way; the Fallback compiler accepts its `Concat` of relabeled branches and rejects duplicate output label sets. Nested aggregation, such as diff --git a/docs/develop_docs/planner-vocabulary-migration.md b/docs/develop_docs/planner-vocabulary-migration.md index f5f7f70ae..ea46e992c 100644 --- a/docs/develop_docs/planner-vocabulary-migration.md +++ b/docs/develop_docs/planner-vocabulary-migration.md @@ -36,8 +36,8 @@ names. | `SketchAlgorithmStrategy::with_models_and_evidence` | `SketchAlgorithmStrategy::new_with_planning_inputs_and_evidence` | | `HydraGroupingStrategy::with_models_and_evidence` | `HydraGroupingStrategy::new_with_planning_inputs_and_evidence` | -For example, `Binder::new().bind(&tree)` becomes -`SchemaResolver::new().resolve_schema(&tree)`. Cost-model implementations that +For example, `Binder::new().bind(&dag)` becomes +`SchemaResolver::new().resolve_schema(&dag)`. Cost-model implementations that accept or return `Implementation` now use `Realization`; variants and ranking contracts remain the same. diff --git a/docs/develop_docs/pre-asap-ir.md b/docs/develop_docs/pre-asap-ir.md index 749f97cf9..dcb3597e2 100644 --- a/docs/develop_docs/pre-asap-ir.md +++ b/docs/develop_docs/pre-asap-ir.md @@ -41,7 +41,7 @@ to one source language. - [`Concat`](#concat) — exact, untyped `UNION ALL` of union-compatible branches. **[PromQL-specific nodes](#promql-specific-nodes)** -- [`PromqlScalarBridge`](#promqlscalarbridge) — a scalar sub-expression at an operator-tree position. +- [`PromqlScalarBridge`](#promqlscalarbridge) — a scalar sub-expression at an operator-DAG position. - [`EvalTimestamp`](#evaltimestamp) — the query evaluation time as a scalar (PromQL `time()`). - [`PromqlVectorFromScalar`](#promqlvectorfromscalar) — promotes a scalar to a label-less instant vector. - [`PromqlScalarFromVector`](#promqlscalarfromvector) — collapses a single-series vector to a scalar. @@ -174,7 +174,7 @@ meaningful summary implementation. `measures[i]`; groups are still formed from every row. It is positional against `child`'s output (like `Filter.pred`), not against the aggregate's output like `having`. Empty means no measure is filtered; that is the only spelling of "unfiltered" a resolved - tree carries, so `[None, None]` is normalized to `[]`. + DAG carries, so `[None, None]` is normalized to `[]`. - `having` — an optional post-aggregation filter predicate (SQL `HAVING`). - `child` — the input being aggregated. @@ -241,7 +241,7 @@ Example for `having`: ) ``` - but the tree can still carry the same condition as a wrapping `Filter` near the root: + but the DAG can still carry the same condition as a wrapping `Filter` near the root: ```text Filter( @@ -329,7 +329,7 @@ the same logical data domain. - `predicates` — row-level filters pushed all the way down to this scan (Rules/Invariants rule 1); enforced structurally at lowering time — a `Filter` directly over a `Scan` never survives. -- `schema` — the binding schema every positional column reference in the tree resolves against. +- `schema` — the binding schema every positional column reference in the DAG resolves against. ### Filter @@ -487,7 +487,7 @@ histogram_quantiles(rate(http_request_duration_seconds_bucket[5m]), "le", 0.5, 0 ### PromqlScalarBridge A scalar sub-expression (issue #220: in practice always `Literal(ScalarValue::Float64(_))` — -a PromQL number literal, or a folded constant scalar expression) sitting at an **operator-tree +a PromQL number literal, or a folded constant scalar expression) sitting at an **operator-DAG position** — a `BinaryOp` operand for ` op ` thresholds and unit conversions, a `PromqlVectorFromScalar` child, or a whole query's root. This wrapper is what marks the position; it no longer duplicates `Literal`'s value the way the old `PromqlScalar(f64)` variant diff --git a/docs/develop_docs/storage-operation-costs.md b/docs/develop_docs/storage-operation-costs.md index 564a6534f..1a8dacdbd 100644 --- a/docs/develop_docs/storage-operation-costs.md +++ b/docs/develop_docs/storage-operation-costs.md @@ -8,7 +8,7 @@ operation counts are unestimated, not inferred to be zero. A supplied profile must cover every reachable physical node, with an explicit empty `accesses` list for nodes doing no storage I/O. Entries bind the complete -`PhysicalDagNode` and `OperatorStatistics`, so a reused ID cannot silently +`PhysicalDAGNode` and `OperatorStatistics`, so a reused ID cannot silently borrow evidence from a different plan. Profiles may contain additional nodes for other alternatives. Their evidence generation must equal the planner snapshot version, and `observed_at_ms <= planning_time < valid_until_ms`. diff --git a/docs/develop_docs/target-candidate-api-migration.md b/docs/develop_docs/target-candidate-api-migration.md index 6139f53dc..a5b081a61 100644 --- a/docs/develop_docs/target-candidate-api-migration.md +++ b/docs/develop_docs/target-candidate-api-migration.md @@ -11,7 +11,7 @@ are unchanged. #453 separately defines the integration API surface. | `SelectedGroup` | `TargetSubDAGSelection` | Selected choice and usage information for one target; the choice may be absent | | `GlobalSelection::groups()` | `GlobalSelection::target_selections()` | Iterate decisions, not alternative sets | | `MaterializeSummaryMaintenanceLifecycleError` | `SummaryMaintenanceLifecycleAssemblyError` | Failure assembling a DAG or deriving maintenance decisions | -| Error variant `Materialize` | `AssembleDag` | Wrap an underlying `RealizationError` from DAG assembly | +| Error variant `Materialize` | `AssembleDAG` | Wrap an underlying `RealizationError` from DAG assembly | | Internal `materialize_inner` / `materialize_residual` | `assemble_target` / `assemble_residual` | Assemble selected nodes, not runtime materialized views | | Internal assembly cache `materialized` | `assembled_nodes` | Preserve shared `Rc` identity | diff --git a/docs/user_guide_docs/run-a-query.md b/docs/user_guide_docs/run-a-query.md index af2cfc07e..3f5825c29 100644 --- a/docs/user_guide_docs/run-a-query.md +++ b/docs/user_guide_docs/run-a-query.md @@ -1,6 +1,6 @@ # ASAPPlanner CLI user guide -Use the `asap-devtools` commands to inspect query IR, export graphs, and inspect +Use the `asap-devtools` commands to inspect query IR, export DAGs, and inspect corpus coverage. These commands do not deploy or execute a physical plan. To develop an application using the Rust library, start with @@ -17,8 +17,8 @@ tool on first use. | --- | --- | --- | | `show_pre_asap_ir --data-ingestion-interval-ms 1000 queries.txt` | File path, or stdin when omitted | Prints canonical Pre-ASAP IR | | `show_post_asap_ir --data-ingestion-interval-ms 1000 queries.txt` | Same query file format | Prints all sketch-strategy Post-ASAP candidates using a fixed approximate target, in cost-model order | -| `dag_export --data-ingestion-interval-ms 1000 --promql ""` | One PromQL expression | Exports a query graph for inspection | -| `dag_export --sql ""` | One SQL expression using the tool's catalog | Exports a query graph for inspection | +| `dag_export --data-ingestion-interval-ms 1000 --promql ""` | One PromQL expression | Exports a query DAG for inspection | +| `dag_export --sql ""` | One SQL expression using the tool's catalog | Exports a query DAG for inspection | | `analyze_corpora --corpora --data-ingestion-interval-ms 1000 --out-dir ` | Repository PromQL corpora, output directory | Writes successful/error IR dumps and summary reports | | `analyze_corpora --sql-corpora --out-dir ` | Repository SQL corpora, output directory | Writes SQL corpus reports | | `variant_coverage --data-ingestion-interval-ms 1000` | Repository corpora | Reports Pre-ASAP IR variant coverage | diff --git a/old_docs/README.md b/old_docs/README.md index d29667e26..0989e65b4 100644 --- a/old_docs/README.md +++ b/old_docs/README.md @@ -18,7 +18,7 @@ are still planned. | Layer | What | Where it lives | |---|---|---| | 1 | Query-language parsing (PromQL, SQL; DataFusion, ElasticDSL planned) | `asap-frontend-promql`, `asap-frontend-sql` | -| 2 | Per-language relational algebra tree + the shared L2→L3 converter (incl. the post-lowering canonicalization pass both languages run through) | `asap-l2` (emitted by the front ends) | +| 2 | Per-language relational algebra DAG + the shared L2→L3 converter (incl. the post-lowering canonicalization pass both languages run through) | `asap-l2` (emitted by the front ends) | | 3 | Intent algebra defined by ASAPPlanner itself — query language and runtime independent IR (intent only — no summary type, no summary params). | `asap-ir::intent_algebra` | | 4 | Cost-aware optimizer — CSE + pluggable cost model + the summary-vs-exact accuracy decision (`AggIntent → SummaryKind`, landed). Produces the **summary-bound** IR (kind + params committed), plus the serving-time `SummaryExecutor` interface that answers a query against it. | binding/optimizer passes in `asap-plan`; summary-bound IR + serving-time executor in `asap-sketch` | | 5 | Physical runtime / Data plane — The planner emits configurations to physical runtime, with the execution environment consider the parallelism, hardware types, lifecycle stages, distributed workers | The implementation of this layer should be in different downstream application repos. | @@ -43,7 +43,7 @@ runtime coupling. Two isolation wins fall out of this: - **The front ends quarantine their parsers** — a caller that needs only PromQL depends on `asap-frontend-promql` and never compiles DataFusion, and vice-versa (verified with `cargo tree`). `asap-lower` is the facade for callers that want both. -- **L3-only consumers skip the L2 machinery** — `asap-sketch` (L4 types) depends on `asap-ir` alone, and `asap-plan` (L3→L4 binding + optimizer) adds only `asap-sketch` on top of that — neither pulls the L2 relational tree, the converter, or the binder (`asap-l2`). Only the front ends, which actually *lower* queries, need `asap-l2`. +- **L3-only consumers skip the L2 machinery** — `asap-sketch` (L4 types) depends on `asap-ir` alone, and `asap-plan` (L3→L4 binding + optimizer) adds only `asap-sketch` on top of that — neither pulls the L2 relational DAG, the converter, or the binder (`asap-l2`). Only the front ends, which actually *lower* queries, need `asap-l2`. ### Directory structure @@ -62,7 +62,7 @@ crates/ │ └── names.rs # BindingName / QueryId ├── l2/ # asap-l2 — L2 relational algebra + L2→L3 converter │ └── src/ -│ ├── relational.rs # L2: per-language relational tree the front ends emit +│ ├── relational.rs # L2: per-language relational DAG the front ends emit │ ├── lower.rs # L2→L3 converter (convert_root) │ ├── canonicalize.rs # shared post-lowering normalization (heavy-hitter TopK, #34) │ ├── binder.rs # positional name-resolution seed @@ -119,7 +119,7 @@ cargo test --workspace one or more `--sql`/`--promql` queries through L1→L2→L3 and flattens each resulting `QueryExpr` — the L3 canonical IR (see [`docs/l2-intent-algebra.md`](docs/l2-intent-algebra.md)) — into a generic -node/edge graph, printed as JSON: +node/edge DAG, printed as JSON: ```bash cargo run -p asap-lower --example dag_export -- \ @@ -130,7 +130,7 @@ cargo run -p asap-lower --example dag_export -- \ `--name` is optional (defaults to `q`); repeat `--sql`/`--promql` to pack several queries into one file. Each node carries a `kind`, a short `label`, its own fields under `detail`, `children` ids, and a bottom-up structural -`hash` so identical subtrees (e.g. a `Scan` shared across two queries) can be +`hash` so identical sub-DAGs (e.g. a `Scan` shared across two queries) can be spotted by comparing hashes. The query above exports as: ```json @@ -138,7 +138,7 @@ spotted by comparing hashes. The query above exports as: "queries": [ { "name": "q1", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "Scan", "label": "Scan(http_requests_total)", "children": [], "detail": { "source": { "TimeSeries": { "metric": "http_requests_total" } }, "predicates": [], "schema": { "...": "..." } } }, { "id": 1, "kind": "TimeRange", "label": "TimeRange(300s)", "children": [0], "detail": { "range": { "secs": 300, "nanos": 0 } } }, @@ -161,4 +161,4 @@ loads) — to browse it as an interactive DAG: click a node for its full `detail` in a side panel, switch between queries via tabs, and toggle highlighting of structurally-identical nodes shared across queries. See [`tools/dag-viewer/README.md`](tools/dag-viewer/README.md) for the shared- -subtree-highlighting caveat (it's a client-side hash proxy, not real CSE). +sub-DAG-highlighting caveat (it's a client-side hash proxy, not real CSE). diff --git a/old_docs/docs/design.md b/old_docs/docs/design.md index 04a78f6d5..0e5ab4ad0 100644 --- a/old_docs/docs/design.md +++ b/old_docs/docs/design.md @@ -27,7 +27,7 @@ query string ▼ per-language AST/DAG, unresolved column references ← L1 │ pass': resolve (bind names to schema positions, substitute them - │ throughout the already-canonical-shaped tree) + │ throughout the already-canonical-shaped DAG) ▼ │ pass'': canonicalize (cross-language / cross-phrasing pattern │ normalization — e.g. promoting a generic @@ -75,7 +75,7 @@ language, or differently-phrased within the same language — converge on the identical canonical shape. `canonicalize` catches the cases where a language has no dedicated syntax for an intent — e.g. SQL's `ORDER BY count DESC LIMIT k` has no `topk()`-shaped AST node; a -pattern-detection pass over the already-assembled tree recognizes it as +pattern-detection pass over the already-assembled DAG recognizes it as the same `Aggregate{aggs:[TopK]}` shape PromQL's dedicated `topk()` produces directly in pass 1. @@ -93,14 +93,14 @@ flowchart LR ## L2 — intent algebra -**Job: define the canonical intent tree's vocabulary — the shape L1's +**Job: define the canonical intent DAG's vocabulary — the shape L1's passes produce — expressed declaratively (e.g. "a quantile to this accuracy," "the top-k by this ranking"), with implementation strategy left to L3.** L2 is the vocabulary/rule set L1's output conforms to, enforced by construction: every front end's output runs through the same `resolve`+`canonicalize`. - The result is a language- and deployment-independent canonical - intent tree: what to compute, without committing to how. Deployment + intent DAG: what to compute, without committing to how. Deployment here refers to a physical execution context — e.g. parallelism and the lifecycle stage a computation runs at — a different sense of "deployment" than the Glossary's "Deployment model" entry below; @@ -111,12 +111,12 @@ output runs through the same `resolve`+`canonicalize`. sub-computations are properties of this canonical form. Both depend on L1's `canonicalize` pass having already converged semantically-equivalent queries onto the same shape: only - structurally-identical sub-trees can be recognized as the same + structurally-identical sub-DAGs can be recognized as the same reusable computation. ```mermaid flowchart LR - L1T["canonical intent tree\n(from L1)"] --> V["intent vocabulary\n+ design rules"] + L1T["canonical intent DAG\n(from L1)"] --> V["intent vocabulary\n+ design rules"] V --> L2G["governs what a valid\nL1 output looks like"] ``` diff --git a/old_docs/docs/l1-query-language.md b/old_docs/docs/l1-query-language.md index 9d2530505..7b3ccd7a1 100644 --- a/old_docs/docs/l1-query-language.md +++ b/old_docs/docs/l1-query-language.md @@ -1,7 +1,7 @@ # L1 — query string → canonical intent algebra Per-language front ends, one per supported query language. Each turns a -raw query string all the way into the canonical intent tree that L2 +raw query string all the way into the canonical intent DAG that L2 (see [`l2-intent-algebra.md`](./l2-intent-algebra.md)) is the vocabulary for. That journey is **two passes**, and this doc is organized around that split (see `design.md`'s representation/pass table for the @@ -36,7 +36,7 @@ language A's native AST language B's native AST │ interpret directly into │ interpret directly into │ the canonical shape │ the canonical shape ▼ ▼ - canonical-shaped tree, named column references + canonical-shaped DAG, named column references ═══════════════ pass 1 done; pass 2 starts ═══════════════ │ one shared pass, for every language: │ 1. resolve — bind names to schema @@ -44,7 +44,7 @@ language A's native AST language B's native AST │ 2. canonicalize — cross-language / │ cross-phrasing normalization ▼ - canonical intent tree ← L1's output, L2's vocabulary + canonical intent DAG ← L1's output, L2's vocabulary columns are positions + a self-contained schema on every scan ``` @@ -52,7 +52,7 @@ language A's native AST language B's native AST ## Pass 1 — interpret Different front ends can look nothing alike where they start — one may -begin from a bare AST, another from an already-planned tree — but every +begin from a bare AST, another from an already-planned DAG — but every front end produces the same canonical shape as its output, through the same node vocabulary, regardless of source language. @@ -66,10 +66,10 @@ to positions is pass 2's job, once a schema exists to resolve against. ### The nesting contract -Nesting — an operator tree inside a source clause or a function +Nesting — an operator DAG inside a source clause or a function argument, an aggregate over an aggregate, a binary operation over two -subtrees, a range function over a sub-query, and so on — is required to -lower **structurally**: the canonical intent tree is a recursive, +sub-DAGs, a range function over a sub-query, and so on — is required to +lower **structurally**: the canonical intent DAG is a recursive, arbitrarily-nestable structure, so "an operation over a sub-query" needs no special-cased IR shape. Any language whose grammar allows nesting must be able to express it this way, using the same recursive @@ -88,7 +88,7 @@ language — does two things in order: 1. **Resolve.** Bind names to schema positions (see "Name resolution: binding" below), then substitute every column reference throughout - the tree — a generic walk over the already-canonical-shaped tree + the DAG — a generic walk over the already-canonical-shaped DAG pass 1 produced. 2. **Canonicalize.** Run a cross-language normalization pass so that semantically equivalent queries, from any supported language or @@ -102,12 +102,12 @@ language — does two things in order: dedicated `topk(k, count_over_time(…))` produces directly in pass 1. This ordering matters: canonicalization operates on an already-resolved -tree, so its pattern-matching rules work over stable schema positions, +DAG, so its pattern-matching rules work over stable schema positions, not per-language surface syntax. ### Name resolution: binding -Binding walks the canonical-shaped, named-reference tree pass 1 +Binding walks the canonical-shaped, named-reference DAG pass 1 produced and derives one self-contained schema that every column reference indexes into: it seeds columns from whatever catalog is available (or a minimal always-present floor if the catalog knows @@ -119,7 +119,7 @@ inline while interpreting, in pass 1: - **Everything downstream becomes purely structural and total.** Take those two words literally: *structural* means later steps match on - tree shape and column position, never on a name string again — the + DAG shape and column position, never on a name string again — the string comparisons all happened once, here. *Total* is the computer-science sense — a function defined for every input, with no case left unhandled — applied to column resolution: once binding has @@ -136,7 +136,7 @@ inline while interpreting, in pass 1: complement is represented and deferred to serving time, rather than resolved eagerly. -Each independent sub-tree (e.g. either side of a binary operation) +Each independent sub-DAG (e.g. either side of a binary operation) binds against its own schema, since the two sides may reference entirely different sources — but a side must still see names referenced by an *enclosing* operation (a grouping key mentioned above, @@ -209,7 +209,7 @@ pub trait SchemaCatalog { pub struct Binder { .. } impl Binder { - pub fn bind(&self, tree: &QueryExpr) -> Schema; + pub fn bind(&self, dag: &QueryExpr) -> Schema; } ``` @@ -225,7 +225,7 @@ with the default: ```rust // PromQL: `sum by (job) (http_requests_total)`, no catalog available. -let schema = Binder::default().bind(&tree); +let schema = Binder::default().bind(&dag); // -> Schema { columns: [ts, value, job], time_index: Some(0), closed: false } // ("job" was seeded because the query references it; anything the // query never mentions is simply absent from this schema) @@ -243,7 +243,7 @@ impl SchemaCatalog for SqlCatalog { .. } } -let schema = Binder::with_catalog(SqlCatalog { .. }).bind(&tree); +let schema = Binder::with_catalog(SqlCatalog { .. }).bind(&dag); // -> Schema { columns: [host, bytes], time_index: None, closed: true } // (the catalog's declared columns are used verbatim, regardless of // which ones the query actually references) @@ -254,10 +254,10 @@ migration tracked in #179, since it already operates on the canonical type either way: ```rust -pub fn canonicalize(tree: QueryExpr) -> QueryExpr; +pub fn canonicalize(dag: QueryExpr) -> QueryExpr; ``` -Idempotent, bottom-up: rewrites a tree already in canonical shape into +Idempotent, bottom-up: rewrites a DAG already in canonical shape into its normal form (e.g. promoting a generic `Limit{Sort{Aggregate([Count])}}` shape to the explicit heavy-hitter `Aggregate{aggs:[TopK]}` form) — same type in, same type out. diff --git a/old_docs/docs/l2-intent-algebra.md b/old_docs/docs/l2-intent-algebra.md index 1b09a7a10..ad280534b 100644 --- a/old_docs/docs/l2-intent-algebra.md +++ b/old_docs/docs/l2-intent-algebra.md @@ -5,9 +5,9 @@ this layer — expressing what to compute; summary type, summary parameters, and physical operator choice are committed later, entirely at L3's discretion (see [`l3-summary-bound-ir.md`](./l3-summary-bound-ir.md)). Data-model -agnostic by design: the same tree shape covers both time-series-style +agnostic by design: the same DAG shape covers both time-series-style sources and tabular sources, through a source's data model living on -its scan node rather than forking the tree type. +its scan node rather than forking the DAG type. ## Design rules @@ -31,13 +31,13 @@ its scan node rather than forking the tree type. 3. **Intent at L2, summary at L3.** An intent like "a quantile to this accuracy" says what to compute; it never says which summary computes it. -4. **A DAG, not a tree.** A producer can have more than one consumer +4. **A DAG with sharing.** A producer can have more than one consumer sharing its computed result — from explicit naming in the source query (e.g. a named sub-query), from a later layer's decision to reuse a shared computation across two different queries, or from physical structure (a pre-aggregate feeding several downstream consumers). The canonical form supports this by construction, via a - binding/reference pair of node kinds — a tree-only representation + binding/reference pair of node kinds — a representation without sharing would have to duplicate the producer per consumer and lose the sharing. @@ -60,10 +60,10 @@ aggregate" as syntactic shapes — see design rule 1; both fold into composition of the generic operators instead, except for the one strategy-driven exception noted there. -**Why a scan's source is a variant, not a separate tree type per data +**Why a scan's source is a variant, not a separate DAG type per data model.** Most operators (filter, aggregate, …) have identical semantics regardless of whether the input is a time-series window or a table scan -— only the leaf differs. One tree with a polymorphic leaf lets +— only the leaf differs. One DAG with a polymorphic leaf lets data-model-agnostic rules apply uniformly across every data model; rules that do care about data-model specifics gate on that leaf's declared kind. Summaries themselves are meant to be data-model-agnostic @@ -104,8 +104,8 @@ unique key, a designated time column — it is a position into a schema, not a string. A pre-bind string name means nothing without knowing which schema it resolves against and at what offset; a position is settled once, at bind time, and by construction can't fail to resolve for any -later pass. This also makes the canonical tree self-describing: every -scan carries its own schema, so any sub-tree's output schema is +later pass. This also makes the canonical DAG self-describing: every +scan carries its own schema, so any sub-DAG's output schema is computable purely from its inputs, without external context. A named alias for the column-identity type (rather than a bare integer) is still kept, so code that touches it can express "this is a column @@ -113,15 +113,15 @@ position" as a distinct kind of value, not just any number. ## Schema flow -Every edge in the canonical tree carries a schema: its columns, which +Every edge in the canonical DAG carries a schema: its columns, which column (if any) is the designated time index, its unique keys, and whether it's closed (a complete enumeration) or open (a runtime row may -carry more). The tree is locally type-checked: a node's output schema +carry more). The DAG is locally type-checked: a node's output schema is a pure function of its inputs' schemas and its own parameters, -verifiable without consulting the rest of the tree. +verifiable without consulting the rest of the DAG. Three distinct kinds of schema-shaped information are easy to conflate -and shouldn't be: the schema flowing along the canonical tree's own +and shouldn't be: the schema flowing along the canonical DAG's own edges; the schema of the underlying data source itself (what tables or metrics actually exist, consulted only during L1's internal name resolution); and a catalog of what summaries exist and what they can @@ -218,7 +218,7 @@ pub enum QueryExpr { ``` `reduction` is a field *on* the `Aggregate` variant itself — not a -separate node in the tree, and not something any other variant carries. +separate node in the DAG, and not something any other variant carries. It answers a question only `Aggregate` ever needs to ask: is this node collapsing rows at all, and if so, by which (possibly empty) key set — or does it have no grouping concept to begin with. Making that an @@ -230,7 +230,7 @@ handling downstream — see [`l3-summary-bound-ir.md`](./l3-summary-bound-ir.md# actually load-bearing. The two small types that field's shape turns on, in full — neither is a -tree node either; both are plain data reachable only through +DAG node either; both are plain data reachable only through `Aggregate.reduction`, and `GroupKeys` only exists at all when `reduction` is `Reduce` (it's meaningless for `PerEntity`, which is exactly why it isn't a sibling field instead): @@ -378,7 +378,7 @@ for its particular statistic, which is exactly the design rule 1 exception (a genuinely different computational access pattern, not an ordinary composition of existing operators). -And the schema every edge in the tree carries: +And the schema every edge in the DAG carries: ```rust pub struct Schema { diff --git a/old_docs/docs/l3-summary-bound-ir.md b/old_docs/docs/l3-summary-bound-ir.md index a86d9b9cd..ae59815e8 100644 --- a/old_docs/docs/l3-summary-bound-ir.md +++ b/old_docs/docs/l3-summary-bound-ir.md @@ -23,7 +23,7 @@ answer it — with no reference to what's actually stored anywhere yet. ### The shape of a summary-bound plan -A summary-bound plan mirrors the canonical intent tree but replaces +A summary-bound plan mirrors the canonical intent DAG but replaces each summarizable node with a decision: - **Unbound / logical** — nothing rewrote this node (e.g. a filter or a @@ -43,11 +43,11 @@ each summarizable node with a decision: - **Summary merge** — combining multiple built summaries into one; only valid when every input agrees on family and parameters. -A summary-bound plan is a DAG, not just a tree — a shared +A summary-bound plan is a DAG with sharing — a shared sub-computation can appear as more than one reference to the same bound node, mirroring the sharing already present in the canonical form (see [`l2-intent-algebra.md`](./l2-intent-algebra.md)). Binding only fires -where an intent is recognizable in the tree; anything underneath an +where an intent is recognizable in the DAG; anything underneath an unrewritten (logical) parent stays logical too — extending `implement` to rewrite through a logical parent is a known open design question, left to a deployment that has a real need for it. @@ -79,17 +79,17 @@ a deployment supplies: how to find candidate materialized instances for a bound node, how to fetch one instance's state, how to merge several instances' states, how to read a value out of a state, and how to evaluate an unrewritten logical node directly. A shared execution -routine walks the bound tree and calls into these deployment-supplied +routine walks the bound DAG and calls into these deployment-supplied operations, so every deployment gets the same walk and merge logic for free. Finding candidates for a summary-aggregation node requires seeing that -node's entire input sub-tree, not just a bare name — a summary +node's entire input sub-DAG, not just a bare name — a summary aggregation node carries no source identity of its own; that identity lives further down, inside a scan. Walking down to find it is deployment-specific knowledge (how a real store names and indexes summarized data); the shared execution routine doesn't need to -interpret it, only pass the sub-tree along. +interpret it, only pass the sub-DAG along. ### Nested composition @@ -186,7 +186,7 @@ reaches serving-time execution: | Node | `summary` field | Constraint | |---|---|---| -| `Logical` | — | none — wraps an arbitrary unrewritten `QueryExpr` subtree | +| `Logical` | — | none — wraps an arbitrary unrewritten `QueryExpr` sub-DAG | | `SummaryAgg` | own | none beyond `col`'s intent already requiring an accuracy target compatible with `summary`; this is the leaf every other row's constraints are checked *against* | | `SummaryJoin` | own | only emitted by a join-specific `Bind*OnJoin` rule — none exist yet in the base design, so this variant has no live producer | | `SummarySubtract` | read from `left`/`right` | `left`/`right` agree on `(kind, params)`; that kind's catalog entry sets `subtractable` | @@ -218,11 +218,11 @@ flowchart LR LG -. "✗ already a value" .-> SM ``` -`implement` — turning a canonical `QueryExpr` into an `L3Node` tree: +`implement` — turning a canonical `QueryExpr` into an `L3Node` DAG: ```rust -pub fn implement_tree(expr: &QueryExpr) -> Result, ImplementError>; -pub fn implement_tree_with(expr: &QueryExpr, cost_model: &dyn CostModel) -> Result, ImplementError>; +pub fn implement_dag(expr: &QueryExpr) -> Result, ImplementError>; +pub fn implement_dag_with(expr: &QueryExpr, cost_model: &dyn CostModel) -> Result, ImplementError>; ``` **"Implementation" is the answer to one question: how is this one @@ -277,7 +277,7 @@ pub trait CostModel { ``` Serving-time — the interface a deployment implements to actually answer -a query against an already-bound `L3Node` tree: +a query against an already-bound `L3Node` DAG: ```rust pub trait SummaryExecutor { diff --git a/old_docs/docs/l4-physical-plan.md b/old_docs/docs/l4-physical-plan.md index 8538ed964..ba8af34d8 100644 --- a/old_docs/docs/l4-physical-plan.md +++ b/old_docs/docs/l4-physical-plan.md @@ -94,7 +94,7 @@ pub trait TopologyDescriptor { // A `StageEdge` names a pair of stages data is allowed to flow // between (e.g. "edge → backend" if that deployment's edge tier // ships summaries up to a backend tier) — the topology's connectivity - // graph, distinct from which stages merely *exist* (`stages()` above). + // DAG, distinct from which stages merely *exist* (`stages()` above). // `StageAllocator::allocate` only assigns a piece of the plan to move // from one stage to another along an edge this list actually // contains; a `TopologyDescriptor` with no edge between two stages is @@ -119,7 +119,7 @@ pub struct Executor { pub address: ExecutorAddr, // OpAMP agent / HTTP endpoint / in-process handle — deployment-defined } -// Stage-level allocation: given an L3 tree + a topology, decide which +// Stage-level allocation: given an L3 DAG + a topology, decide which // nodes land on which stage, subject to deployment constraints. // Per-executor fan-out is the deployment model's own PhysicalPlanner, // using the executor list from DeploymentConstraints::executors(). diff --git a/tools/dag-viewer/README.md b/tools/dag-viewer/README.md index d2c1a65ff..b90aae272 100644 --- a/tools/dag-viewer/README.md +++ b/tools/dag-viewer/README.md @@ -72,7 +72,7 @@ physical-evidence document: an immutable `evidence_version`, calibration, and target records containing the exact target `QueryExpr` and comparison scope. Each exact replacement candidate owns its complete logical-node `PhysicalNodeEvidence`; summary candidates additionally own their bound -`PhysicalDag`. Candidate-local evidence prevents statistics for one physical +`PhysicalDAG`. Candidate-local evidence prevents statistics for one physical alternative from satisfying another. Candidate matching includes the complete exported plan, including accuracy guarantees, and never uses a hash or strategy name; derived floating constants allow only a one-ULP JSON round-trip tolerance. @@ -93,13 +93,13 @@ cargo run -p asap-devtools --bin dag_export -- \ It ranks candidates with the planner's structural `DefaultCostModel`, so the structure of the export is real — which replacements the search found, which -one won per group, and the merged post-ASAP graph — while no cost is exported +one won per group, and the merged post-ASAP DAG — while no cost is exported at all. Every `CostAnnotation` stays `Unavailable` with no `value` and renders as **Not estimated**; the structural ranking number is never serialized. Use it to see what ASAPPlanner does with a workload before there is a deployment to calibrate against, and `--planner-cost-json` once there is. -Without either flag, `--post-asap` exports the raw graph only. +Without either flag, `--post-asap` exports the raw DAG only. The viewer also accepts the JSON produced by `export_summary_maintenance_plan`. It renders the materialized summary DAG as @@ -121,7 +121,7 @@ always opens in Pre/Post-ASAP mode; there is no `--mode` option. ## JSON contract -`NamedGraph.graph` is the original pre-ASAP DAG. `NamedGraph.post_graph` +`NamedDAG.dag` is the original pre-ASAP DAG. `NamedDAG.post_dag` is the complete translated DAG. Every post-ASAP node produced or carried by a selected replacement directly contains: @@ -184,14 +184,14 @@ operation, `1e-10` per scan byte, and `1e-9` per peak-memory byte, the displayed not statistics inferred by the viewer. The same three fields also appear on `TargetReplacement` -(replacement-region baseline/selected/benefit), `NamedGraph.workload_cost` / -`WorkloadGraph.workload_cost` (whole selected-workload cost/benefit, shared -decisions counted once via `decision.id` dedup). `DagGraph.edge_annotations` +(replacement-region baseline/selected/benefit), `NamedDAG.workload_cost` / +`WorkloadDAG.workload_cost` (whole selected-workload cost/benefit, shared +decisions counted once via `decision.id` dedup). `ExportDAG.edge_annotations` is reserved for a higher layer that has physical evidence for a particular -edge; graph sharing alone never creates an edge cost. The sidebar shows the full breakdown +edge; DAG sharing alone never creates an edge cost. The sidebar shows the full breakdown (value, unit, provenance, baseline, ratio, inputs) on node/edge click and in the workload-scope summary; a post-ASAP node with a costed decision also -gets a concise on-graph `▼NN%`/`▲NN%` badge next to its label. +gets a concise on-DAG `▼NN%`/`▲NN%` badge next to its label. All of this is additive and optional: an export with none of these fields (anything produced before issue #286) renders exactly as before. diff --git a/tools/dag-viewer/dag.example.json b/tools/dag-viewer/dag.example.json index da3e88708..85c0e97b6 100644 --- a/tools/dag-viewer/dag.example.json +++ b/tools/dag-viewer/dag.example.json @@ -3,7 +3,7 @@ { "name": "q1", "source": "SELECT service, COUNT(*) FROM metrics GROUP BY service", - "graph": { + "dag": { "nodes": [ { "id": 0, @@ -361,14 +361,14 @@ }, "after": { "kind": "Summary", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": { - "pre_asap_subgraph": { + "pre_asap_sub_dag": { "nodes": [ { "children": [], @@ -564,7 +564,7 @@ } } ], - "post_graph": { + "post_dag": { "nodes": [ { "id": 0, diff --git a/tools/dag-viewer/index.html b/tools/dag-viewer/index.html index 62ffd88f2..17754d280 100644 --- a/tools/dag-viewer/index.html +++ b/tools/dag-viewer/index.html @@ -447,7 +447,7 @@

ASAP query DAG viewer

-
Drop WorkloadGraph JSON here, or click to choose file(s)
+
Drop WorkloadDAG JSON here, or click to choose file(s)
@@ -456,7 +456,7 @@

ASAP query DAG viewer

Click an edge to inspect its schema
diff --git a/tools/dag-viewer/node-style.js b/tools/dag-viewer/node-style.js index 447f08801..a957817c2 100644 --- a/tools/dag-viewer/node-style.js +++ b/tools/dag-viewer/node-style.js @@ -112,7 +112,7 @@ const CATEGORIES = { // Loud fallback for malformed or version-skewed exports. unknown: { label: 'Unknown kind', - description: 'A DagNode.kind with no KIND_CATEGORY entry — update node-style.js', + description: 'A DAGNode.kind with no KIND_CATEGORY entry — update node-style.js', light: { bg: '#fef2f2', border: '#b91c1c' }, dark: { bg: '#2a1212', border: '#f87171' }, }, @@ -120,7 +120,7 @@ const CATEGORIES = { for (const [kind, category] of Object.entries(KIND_CATEGORY)) { if (!Object.prototype.hasOwnProperty.call(CATEGORIES, category)) { - throw new Error(`DagNode kind ${kind} uses undeclared category ${category}`); + throw new Error(`DAGNode kind ${kind} uses undeclared category ${category}`); } } @@ -139,7 +139,7 @@ function categoryOf(kind) { const category = KIND_CATEGORY[kind]; if (category) return category; console.warn( - `node-style.js: DagNode.kind ${JSON.stringify(kind)} has no KIND_CATEGORY entry — ` + + `node-style.js: DAGNode.kind ${JSON.stringify(kind)} has no KIND_CATEGORY entry — ` + 'rendering as "Unknown kind" instead of silently guessing. Add an entry to KIND_CATEGORY.' ); return 'unknown'; diff --git a/tools/dag-viewer/post_asap_fixture.json b/tools/dag-viewer/post_asap_fixture.json index a680d28f4..ea2de185f 100644 --- a/tools/dag-viewer/post_asap_fixture.json +++ b/tools/dag-viewer/post_asap_fixture.json @@ -3,7 +3,7 @@ { "name": "p95_pktlen", "source": "SELECT srcip, approx_percentile_cont(pkt_len, 0.95) AS p95_pkt_len FROM netflow_table GROUP BY srcip", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {"source": {"Table": {"table_ref": "netflow_table"}}}, "children": [], "hash": 111 }, { "id": 1, "kind": "Aggregate", "label": "Aggregate(1 measures)", "detail": {"measures": [{"kind": "quantile", "q": 0.95, "col": 6, "accuracy": {"Epsilon": 0.01}}], "output_names": ["approx_percentile_cont(netflow_table.pkt_len,Float64(0.95))"], "reduction": {"Reduce": [1]}}, "children": [0], "hash": 222, @@ -12,11 +12,11 @@ ], "root": 2 }, - "post_graph": { + "post_dag": { "nodes": [ - { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {"pre_asap_subgraph": {"nodes": [{"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": []}], "root": 0}}, "children": [] }, + { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {"pre_asap_sub_dag": {"nodes": [{"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": []}], "root": 0}}, "children": [] }, { "id": 1, "kind": "SummaryAgg", "label": "SummaryAgg(Kll)", "detail": {"family": {"Sketch": ["Kll", {"k": 200}]}, "col": {"Column": 6}, "reduction": {"Reduce": [1]}, "grouping": "PerSubpopulationInstance"}, "children": [0], "origin_pre_id": 1 }, - { "id": 2, "kind": "KeepPreAsap", "label": "KeepPreAsap(Project)", "detail": {"pre_asap_subgraph": {"nodes": [{"id": 0, "kind": "Project", "label": "Project(2 cols)", "detail": {}, "children": []}], "root": 0}}, "children": [1] } + { "id": 2, "kind": "KeepPreAsap", "label": "KeepPreAsap(Project)", "detail": {"pre_asap_sub_dag": {"nodes": [{"id": 0, "kind": "Project", "label": "Project(2 cols)", "detail": {}, "children": []}], "root": 0}}, "children": [1] } ], "root": 2 }, @@ -37,9 +37,9 @@ }, "after": { "kind": "Summary", - "graph": { + "dag": { "nodes": [ - { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {"pre_asap_subgraph": {"nodes": [{"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": [], "hash": 111}], "root": 0}}, "children": [] }, + { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {"pre_asap_sub_dag": {"nodes": [{"id": 0, "kind": "Scan", "label": "Scan(netflow_table)", "detail": {}, "children": [], "hash": 111}], "root": 0}}, "children": [] }, { "id": 1, "kind": "SummaryAgg", "label": "SummaryAgg(Kll)", "detail": {"family": {"Sketch": ["Kll", {"k": 200}]}, "col": {"Column": 6}, "reduction": {"Reduce": [1]}, "grouping": "PerSubpopulationInstance"}, "children": [0], "origin_pre_id": 1 } ], "root": 1 @@ -62,7 +62,7 @@ }, "after": { "kind": "Summary", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "KeepPreAsap", "label": "KeepPreAsap(Scan)", "detail": {}, "children": [] }, { "id": 1, "kind": "SummaryAgg", "label": "SummaryAgg(HydraKll)", "detail": {"family": {"Sketch": ["Kll", {"k": 200}]}, "grouping": {"SharedMultiSubpopulation": {"params": {}}}}, "children": [0], "origin_pre_id": 1 } @@ -76,7 +76,7 @@ { "name": "avg_latency", "source": "SELECT service, AVG(latency) FROM metrics GROUP BY service", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "Scan", "label": "Scan(metrics)", "detail": {"source": {"Table": {"table_ref": "metrics"}}}, "children": [], "hash": 444 }, { "id": 1, "kind": "Aggregate", "label": "Aggregate(1 measures)", "detail": {"measures": [{"kind": "avg", "col": 3}], "output_names": ["avg(metrics.latency)"]}, "children": [0], "hash": 555 }, @@ -101,7 +101,7 @@ }, "after": { "kind": "Rewrite", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "Scan", "label": "Scan(metrics)", "detail": {}, "children": [], "hash": 444 }, { "id": 1, "kind": "Aggregate", "label": "Aggregate(2 measures)", "detail": {"measures": [{"kind": "sum", "col": 3}, {"kind": "count"}], "output_names": ["sum(metrics.latency)", "count(*)"]}, "children": [0], "hash": 777 }, @@ -116,7 +116,7 @@ { "name": "cse_share_demo", "source": "-- q_a -- SELECT service, COUNT(*) FROM metrics GROUP BY service ;\n-- q_b -- SELECT service, COUNT(*) FROM metrics WHERE region='us' GROUP BY service", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "Scan", "label": "Scan(metrics)", "detail": {}, "children": [], "hash": 111 }, { "id": 1, "kind": "Aggregate", "label": "Aggregate(1 measures)", "detail": {"measures": [{"kind": "count"}]}, "children": [0], "hash": 999 } @@ -126,7 +126,7 @@ "replacements": [ { "target_pre_id": 0, - "strategy": "SharedSubtree", + "strategy": "SharedSubDAG", "provenance": "CseShare", "rationale": "Scan(metrics) has 2 consumers across this workload — building it once and sharing the Rc beats recomputing it independently per consumer at this estimated row count.", "rank": 0, @@ -137,7 +137,7 @@ }, "after": { "kind": "Rewrite", - "graph": { + "dag": { "nodes": [ { "id": 0, "kind": "Scan", "label": "Scan(metrics) [shared, 2 consumers]", "detail": {}, "children": [], "hash": 111 } ], "root": 0 } diff --git a/tools/dag-viewer/render.py b/tools/dag-viewer/render.py index 6659561ef..b131bbc9f 100755 --- a/tools/dag-viewer/render.py +++ b/tools/dag-viewer/render.py @@ -1,5 +1,5 @@ #!/usr/bin/env python3 -"""Bake WorkloadGraph JSON into a single, portable Pre/Post-ASAP HTML page, +"""Bake WorkloadDAG JSON into a single, portable Pre/Post-ASAP HTML page, with the same query selection, workload-union, and node details as index.html, but with the vendored JS libraries and the query data all inlined into one file. @@ -19,13 +19,13 @@ comment) and only differs in packaging: one query's worth of exported `QueryExpr` detail *is* its plan (see the side panel on node click), and shared-hash highlighting *is* what this repo has for CSE today — both a -hash-based proxy, not real CSE output; see README.md's "Shared-subtree +hash-based proxy, not real CSE output; see README.md's "Shared-sub-DAG highlighting is a proxy" section. Structured cost/benefit annotations (issue #286, `CostAnnotation` in crates/types/src/cost.rs) pass through this deep-copy untouched, same as everything else `prepare_workload` below doesn't explicitly rewrite — viewer.js reads `decision.baseline_cost` / -`.selected_cost` / `.benefit`, `NamedGraph.workload_cost`, and -`DagGraph.edge_annotations` directly, with no help needed from this file. +`.selected_cost` / `.benefit`, `NamedDAG.workload_cost`, and +`ExportDAG.edge_annotations` directly, with no help needed from this file. Usage: cargo run -p asap-devtools --bin dag_export -- --sql "..." --name q1 \\ @@ -113,7 +113,7 @@ def _measure(value: object, input_schema: object = None) -> str: def _bounded_lines(lines: list[str], maximum: int = 4, width: int = 64) -> str: - """Keep graph boxes scannable; the sidebar owns the lossless detail.""" + """Keep DAG boxes scannable; the sidebar owns the lossless detail.""" shortened = [line if len(line) <= width else line[: width - 1] + "…" for line in lines] if len(shortened) > maximum: shortened = shortened[: maximum - 1] + [f"… +{len(shortened) - maximum + 1} more"] @@ -213,30 +213,30 @@ def _semantic_label(node: dict, input_schema: object = None) -> str: elif kind == "SummaryDelete": lines.append(f"key: {_compact(detail.get('key'))}") elif kind == "KeepPreAsap": - nested = detail.get("pre_asap_subgraph") + nested = detail.get("pre_asap_sub_dag") nested_nodes = nested.get("nodes", []) if isinstance(nested, dict) else [] nested_root = nested.get("root") if isinstance(nested, dict) else None root = next((item for item in nested_nodes if item.get("id") == nested_root), None) - lines.append(f"unchanged: {root.get('kind', 'pre-ASAP subtree') if root else 'pre-ASAP subtree'}") + lines.append(f"unchanged: {root.get('kind', 'pre-ASAP sub-DAG') if root else 'pre-ASAP sub-DAG'}") else: # Less common variants still show their own scalar IR fields. Avoid - # schema/subgraph blobs, which belong in the click-to-inspect panel. + # schema/sub-DAG blobs, which belong in the click-to-inspect panel. for key, value in detail.items(): - if key not in {"schema", "pre_asap_subgraph"} and value not in (None, [], {}): + if key not in {"schema", "pre_asap_sub_dag"} and value not in (None, [], {}): lines.append(f"{key}: {_compact(value)}") return _bounded_lines(lines) def prepare_workload(workload: dict) -> dict: - """Copy a workload and replace every graph label with readable IR text.""" + """Copy a workload and replace every DAG label with readable IR text.""" prepared = copy.deepcopy(workload) - def prepare_graph(graph: object) -> None: - if not isinstance(graph, dict): + def prepare_dag(dag: object) -> None: + if not isinstance(dag, dict): return - by_id = {node.get("id"): node for node in graph.get("nodes", [])} - for node in graph.get("nodes", []): + by_id = {node.get("id"): node for node in dag.get("nodes", [])} + for node in dag.get("nodes", []): child = by_id.get((node.get("children") or [None])[0]) input_schema = ( child.get("schema") or (child.get("detail") or {}).get("schema") @@ -244,17 +244,17 @@ def prepare_graph(graph: object) -> None: else None ) node["label"] = _semantic_label(node, input_schema) - nested = (node.get("detail") or {}).get("pre_asap_subgraph") - prepare_graph(nested) + nested = (node.get("detail") or {}).get("pre_asap_sub_dag") + prepare_dag(nested) for query in prepared.get("queries", []): - prepare_graph(query.get("graph")) - prepare_graph(query.get("post_graph")) + prepare_dag(query.get("dag")) + prepare_dag(query.get("post_dag")) return prepared def load_workload(paths: list[Path]) -> dict: - """Merge one or more WorkloadGraph JSON files into one, matching + """Merge one or more WorkloadDAG JSON files into one, matching viewer.js's loadFiles(): a query name colliding with an earlier one is disambiguated by suffixing the source filename.""" queries = [] @@ -262,11 +262,11 @@ def load_workload(paths: list[Path]) -> dict: for path in paths: data = json.loads(path.read_text()) incoming = data.get("queries", []) - if not incoming and isinstance(data.get("graph"), dict) and isinstance(data.get("deployments"), list): + if not incoming and isinstance(data.get("dag"), dict) and isinstance(data.get("deployments"), list): incoming = [{ "name": path.stem or "Summary maintenance plan", - "graph": data["graph"], - "post_graph": data["graph"], + "dag": data["dag"], + "post_dag": data["dag"], "lifecycle_plan": True, "lifecycle_summary": { "selected_raw_recompute": data.get("selected_raw_recompute", False), @@ -291,7 +291,7 @@ def load_workload(paths: list[Path]) -> dict: def _json_script(obj: object) -> str: """JSON-serialize `obj` for embedding inside an HTML ') def test_embedded_workload_round_trips(self): - workload = {"queries": [named_graph("q1"), named_graph("q2")]} + workload = {"queries": [named_dag("q1"), named_dag("q2")]} html = render(workload) m = re.search( @@ -177,7 +177,7 @@ def test_embedded_workload_round_trips(self): def test_standalone_export_carries_the_bulk_selection_control(self): """A generated page gets Select-all for free: render.py inlines the markup and viewer.js verbatim, so neither fix needs its own step.""" - html = render({"queries": [named_graph("q1"), named_graph("q2")]}) + html = render({"queries": [named_dag("q1"), named_dag("q2")]}) self.assertIn('id="selectAllToggle"', html) self.assertIn("function bulkSelectionState(", html) # And it must stay a selection control, not a second Clear all: the @@ -190,14 +190,14 @@ def test_standalone_export_carries_the_bulk_selection_control(self): def test_standalone_export_lanes_are_pannable(self): """Dragging the lane background pans the viewport in the generated page too -- `grabbable: false` alone made it a dead zone.""" - html = render({"queries": [named_graph("q1")]}) + html = render({"queries": [named_dag("q1")]}) lanes = re.findall(r"classes: 'laneParent'[^}]*}", html) self.assertEqual(len(lanes), 2, "expected the union and single-query lanes") for lane in lanes: self.assertIn("pannable: true", lane) def test_render_does_not_mutate_callers_workload(self): - workload = {"queries": [named_graph("q1")]} + workload = {"queries": [named_dag("q1")]} original = json.loads(json.dumps(workload)) render(workload) self.assertEqual(workload, original) @@ -207,7 +207,7 @@ def test_render_adds_no_legacy_mode_config(self): # (which documents the window.__DAG_RENDER__ config object in prose) # once that file is inlined verbatim -- match the actual assignment # statement instead. - workload = {"queries": [named_graph("q1"), named_graph("q2")]} + workload = {"queries": [named_dag("q1"), named_dag("q2")]} html = render(workload) self.assertNotRegex(html, r"window\.__DAG_RENDER__ =") @@ -216,14 +216,14 @@ def test_embedded_data_placed_before_viewer_js_body(self): # synchronously as soon as it runs, so it must appear earlier in the # document than viewer.js's own inlined # " substring must not prematurely close the embedded # '")]} + workload = {"queries": [named_dag("q1", source="SELECT ''")]} html = render(workload) m = re.search( @@ -293,7 +293,7 @@ def test_sort_names_expression_direction_and_null_order(self): } self.assertEqual(_semantic_label(node), "Sort\nsort: col[2] descending, nulls first") - def test_prepares_before_after_and_whole_post_asap_graphs(self): + def test_prepares_before_after_and_whole_post_asap_dags(self): node = { "id": 0, "kind": "Aggregate", @@ -301,22 +301,22 @@ def test_prepares_before_after_and_whole_post_asap_graphs(self): "detail": {"measures": [{"kind": "avg", "col": 3}]}, "children": [], } - def graph(): + def dag(): return {"nodes": [dict(node)], "root": 0} workload = { "queries": [ { - "graph": graph(), - "post_graph": graph(), - "replacements": [{"before": graph(), "after": {"graph": graph()}}], + "dag": dag(), + "post_dag": dag(), + "replacements": [{"before": dag(), "after": {"dag": dag()}}], } ] } prepared = prepare_workload(workload) query = prepared["queries"][0] labels = [ - query["graph"]["nodes"][0]["label"], - query["post_graph"]["nodes"][0]["label"], + query["dag"]["nodes"][0]["label"], + query["post_dag"]["nodes"][0]["label"], ] self.assertEqual(labels, ["Aggregate\nmeasure: avg(col[3])"] * 2) self.assertEqual( @@ -324,13 +324,13 @@ def graph(): "Aggregate(1 measures)", ) - def test_replacement_subgraphs_are_not_prepared_for_the_current_viewer(self): - def graph(): + def test_replacement_sub_dags_are_not_prepared_for_the_current_viewer(self): + def dag(): return {"nodes": [{"id": 0, "kind": "Scan", "label": "legacy", "detail": {}, "children": []}], "root": 0} - workload = {"queries": [{"graph": graph(), "replacements": [{"before": graph(), "after": {"graph": graph()}}]}]} + workload = {"queries": [{"dag": dag(), "replacements": [{"before": dag(), "after": {"dag": dag()}}]}]} replacement = prepare_workload(workload)["queries"][0]["replacements"][0] self.assertEqual(replacement["before"]["nodes"][0]["label"], "legacy") - self.assertEqual(replacement["after"]["graph"]["nodes"][0]["label"], "legacy") + self.assertEqual(replacement["after"]["dag"]["nodes"][0]["label"], "legacy") class MainCliTests(unittest.TestCase): @@ -381,7 +381,7 @@ def annotation(value, cache): result["cache_profile"] = cache return result - return {"sourceBatch": batch, "post_graph": {"nodes": [{"decision": { + return {"sourceBatch": batch, "post_dag": {"nodes": [{"decision": { "id": 1, "baseline_cost": annotation(10, profile), "selected_cost": annotation(4, selected_profile or profile), @@ -438,7 +438,7 @@ def test_empty_render_resets_bulk_selection(self): def test_cache_profile_and_inputs_are_rendered(self): """The sidebar exposes the assumptions behind a cache-adjusted cost.""" - annotation = self.query("warm-cache-v1")["post_graph"]["nodes"][0]["decision"]["baseline_cost"] + annotation = self.query("warm-cache-v1")["post_dag"]["nodes"][0]["decision"]["baseline_cost"] annotation["inputs"] = [{"name": "result_cache_hit_ratio", "value": 0.5, "unit": "ratio"}] html = self.js.call("renderCostAnnotation", "Baseline", annotation) self.assertIn("cache warm-cache-v1", html) diff --git a/tools/dag-viewer/viewer.js b/tools/dag-viewer/viewer.js index 62b2e84ca..1db3e835d 100644 --- a/tools/dag-viewer/viewer.js +++ b/tools/dag-viewer/viewer.js @@ -8,20 +8,20 @@ cytoscape.use(window.cytoscapeDagre); // ── State ──────────────────────────────────────────────────────────────── -// queries: [{ name, graph: { nodes, root }, replacements?, post_graph? }], +// queries: [{ name, dag: { nodes, root }, replacements?, post_dag? }], // flattened across every loaded file (a later file whose query name // collides with an earlier one is kept distinct by suffixing the file // index). `replacements` is the optional --post-asap array of // ReplacementSite entries for that query (each with its own `before`/`after` -// subtree) — defaulted to `[]` when the loaded JSON omits the field, so -// callers never need an extra existence check. `post_graph` is the optional -// --post-asap whole-query merged post-ASAP graph (same flattened -// `{nodes, root}` shape as `graph`, but nodes may be post-ASAP-only kinds +// sub-DAG) — defaulted to `[]` when the loaded JSON omits the field, so +// callers never need an extra existence check. `post_dag` is the optional +// --post-asap whole-query merged post-ASAP DAG (same flattened +// `{nodes, root}` shape as `dag`, but nodes may be post-ASAP-only kinds // like "SummaryAgg" mixed in, and any such node has no `hash` — there's no // corresponding QueryExpr to hash) — left `undefined` when absent (omitted // whenever --post-asap wasn't set, or this query had zero replacements), // unlike `replacements` which always defaults to an array. `workload_cost` -// is the optional per-query `NamedGraph.workload_cost` (issue #286), also +// is the optional per-query `NamedDAG.workload_cost` (issue #286), also // left `undefined` when absent. `sourceBatch` is a viewer-assigned integer // (never present in the JSON itself) shared by every query loaded from the // same document — see computeSelectionWorkloadCost's own doc for why it @@ -36,7 +36,7 @@ let zoom = 1; let participants = new Set(); // Every query pushed from the *same* loaded JSON document (one `dag_export` // process invocation) shares one `sourceBatch` id, assigned here. Needed -// because `DagDecision.id` is only unique *within* one dag_export run, not +// because `DAGDecision.id` is only unique *within* one dag_export run, not // across independently-generated files — computeSelectionWorkloadCost below // dedups by `${sourceBatch}:${decision.id}`, never `decision.id` alone, so // two files that happen to reuse the same small integer id never collide. @@ -122,10 +122,10 @@ function loadFiles(fileList) { reader.onload = () => { try { const parsed = JSON.parse(reader.result); - const incoming = parsed.queries || (parsed.graph && parsed.deployments ? [{ + const incoming = parsed.queries || (parsed.dag && parsed.deployments ? [{ name: file.name.replace(/\.json$/i, '') || 'Summary maintenance plan', - graph: parsed.graph, - post_graph: parsed.graph, + dag: parsed.dag, + post_dag: parsed.dag, lifecycle_plan: true, lifecycle_summary: lifecyclePlanSummary(parsed), }] : []); @@ -137,7 +137,7 @@ function loadFiles(fileList) { let name = q.name; if (existingNames.has(name)) name = `${q.name} (${file.name})`; existingNames.add(name); - queries.push({ name, graph: q.graph, source: q.source, replacements: q.replacements || [], post_graph: q.post_graph, workload_cost: q.workload_cost, lifecycle_plan: q.lifecycle_plan, lifecycle_summary: q.lifecycle_summary, sourceBatch }); + queries.push({ name, dag: q.dag, source: q.source, replacements: q.replacements || [], post_dag: q.post_dag, workload_cost: q.workload_cost, lifecycle_plan: q.lifecycle_plan, lifecycle_summary: q.lifecycle_summary, sourceBatch }); }); } catch (err) { alert(`Failed to parse ${file.name}: ${err.message}`); @@ -417,7 +417,7 @@ function buildCyStyle() { // Default layout animates dagre's computed positions in rather than snapping // to them — every mode switch, checkbox toggle, and file load rebuilds `cy` // from scratch (see the `cy.destroy()` below), so this animation is what -// makes the graph read as "assembling" instead of flickering to a new +// makes the DAG read as "assembling" instead of flickering to a new // static image each time. `elements` also start at opacity 0 and fade in // (paired with the 'node'/'edge' transition-property set in buildCyStyle) // so a freshly-added element doesn't just pop in mid-layout-animation. @@ -442,7 +442,7 @@ function buildCy(elements, layout) { } // Root borders and click/tap wiring for the Pre/Post-ASAP lanes. -function finalizeGraphInteractions() { +function finalizeDAGInteractions() { // Use borders rather than pictograms so IR text owns the whole node box. cy.nodes('[?root]').addClass('root'); @@ -487,13 +487,13 @@ function renderPrePostAsap() { showModeHint('Select one query for a complete pre/post DAG, or several queries for a batch workload view.'); return; } - const missing = selected.filter((q) => !q.post_graph); + const missing = selected.filter((q) => !q.post_dag); if (missing.length > 0) { viewTitleEl.textContent = selected.length === 1 ? `Pre/Post-ASAP: ${selected[0].name}` : `Pre/Post-ASAP workload: ${selected.length} queries`; const action = document.getElementById('plannerRun') ? 'Open Query planner and click “Plan selected workload”, or re-export with dag_export --post-asap.' : 'Re-export with dag_export --post-asap.'; - showModeHint(`No post-ASAP graph for: ${missing.map((q) => q.name).join(', ')}. ${action}`); + showModeHint(`No post-ASAP DAG for: ${missing.map((q) => q.name).join(', ')}. ${action}`); return; } hideModeHint(); @@ -502,12 +502,12 @@ function renderPrePostAsap() { const elements = laneElements( 'summary-maintenance', `${selected[0].name} · lifecycle plan`, - selected[0].post_graph, + selected[0].post_dag, selected[0], 'post', ); buildCy(elements); - finalizeGraphInteractions(); + finalizeDAGInteractions(); applyHighlighting(); const initial = cy.nodes().filter((node) => !node.data('isLane') && node.data('root')).first(); if (initial && initial.length) { @@ -526,8 +526,8 @@ function renderPrePostAsap() { const costSummary = selected.length === 1 ? selected[0].workload_cost : computeSelectionWorkloadCost(selected); const elements = selected.length === 1 ? [ - ...laneElements('pre-asap', `${selected[0].name} · pre-ASAP`, selected[0].graph, selected[0], 'pre', costSummary && costSummary.baseline_cost), - ...laneElements('post-asap', `${selected[0].name} · post-ASAP`, selected[0].post_graph, selected[0], 'post', costSummary && costSummary.selected_cost), + ...laneElements('pre-asap', `${selected[0].name} · pre-ASAP`, selected[0].dag, selected[0], 'pre', costSummary && costSummary.baseline_cost), + ...laneElements('post-asap', `${selected[0].name} · post-ASAP`, selected[0].post_dag, selected[0], 'post', costSummary && costSummary.selected_cost), ] : [ ...unionStageLaneElements('pre', chosen, costSummary && costSummary.baseline_cost), @@ -535,7 +535,7 @@ function renderPrePostAsap() { ]; buildCy(elements); - finalizeGraphInteractions(); + finalizeDAGInteractions(); applyHighlighting(); const initial = cy.nodes().filter((node) => !node.data('isLane') && node.data('stage') === 'pre' && node.data('root') @@ -553,13 +553,13 @@ function unionStageLaneElements(stage, chosen, laneCost) { // Workload merging is an exporter decision. `workload_node_id` is the // explicit JSON mapping; do not reconstruct identity from node content. const laneId = `${stage}-asap-union`; - const graphFor = (query) => (stage === 'pre' ? query.graph : query.post_graph); + const dagFor = (query) => (stage === 'pre' ? query.dag : query.post_dag); const owners = new Map(); chosen.forEach((qIdx) => { const query = queries[qIdx]; - const graph = graphFor(query); - graph.nodes.forEach((node) => { + const dag = dagFor(query); + dag.nodes.forEach((node) => { if (node.workload_node_id === undefined) return; if (!owners.has(node.workload_node_id)) owners.set(node.workload_node_id, new Set()); owners.get(node.workload_node_id).add(query.name); @@ -579,16 +579,16 @@ function unionStageLaneElements(stage, chosen, laneCost) { // viewport. `grabbable: false` alone made it a dead zone — cytoscape // begins a pan only when the element under the pointer is `pannable()`, // which defaults to false for nodes, and the lane covers most of the - // canvas once the graph is zoomed past the viewport. Still not grabbable: + // canvas once the DAG is zoomed past the viewport. Still not grabbable: // cytoscape overrides `grabbable` for a pannable node. { data: { id: laneId, label: laneCostLabel(`${stage === 'pre' ? 'pre-ASAP' : 'post-ASAP'} · workload union`, laneCost, stage), isLane: true }, classes: 'laneParent', selectable: false, grabbable: false, pannable: true }, ]; chosen.forEach((qIdx) => { const query = queries[qIdx]; - const graph = graphFor(query); - const byId = new Map(graph.nodes.map((node) => [node.id, node])); - graph.nodes.forEach((node) => { + const dag = dagFor(query); + const byId = new Map(dag.nodes.map((node) => [node.id, node])); + dag.nodes.forEach((node) => { const key = keyFor(qIdx, node); let entry = entries.get(key); if (!entry) { @@ -599,7 +599,7 @@ function unionStageLaneElements(stage, chosen, laneCost) { for (const decision of translationsForNode(query, node, stage)) { entry.decisions.set(decision.id, decision); } - if (node.id === graph.root) entry.rootFor.add(query.name); + if (node.id === dag.root) entry.rootFor.add(query.name); node.children.forEach((childId) => { const childKey = keyFor(qIdx, byId.get(childId)); const child = byId.get(childId); @@ -636,18 +636,18 @@ function unionStageLaneElements(stage, chosen, laneCost) { // Builds one Pre/Post-ASAP lane (a dashed compound parent plus its // nodes/edges). `nodes` is either a plain pre-ASAP -// DagNode list (the `before` subtree, or an `after.kind === "Rewrite"` -// graph) or a SummaryDagNode list (an `after.kind === "Summary"` graph) — +// DAGNode list (the `before` sub-DAG, or an `after.kind === "Rewrite"` +// DAG) or a SummaryDAGNode list (an `after.kind === "Summary"` DAG) — // both shapes carry id/kind/label/detail/children, which is all a lane -// needs; SummaryDagNode's missing `hash`/`notes` fields are simply never +// needs; SummaryDAGNode's missing `hash`/`notes` fields are simply never // read by this function or by showPrePostDetail below. -function laneElements(laneId, laneLabel, graph, query, stage, laneCost) { - const nodes = graph.nodes; +function laneElements(laneId, laneLabel, dag, query, stage, laneCost) { + const nodes = dag.nodes; const byId = new Map(nodes.map((node) => [node.id, node])); // Keep the delimiter escaped in source. A literal NUL is replaced by the // HTML tokenizer when viewer.js is embedded by render.py, which otherwise // makes these keys differ from the lookup below. - const edgeCostByPair = new Map((graph.edge_annotations || []).map((edge) => [`${edge.from}\u0000${edge.to}`, edge.cost])); + const edgeCostByPair = new Map((dag.edge_annotations || []).map((edge) => [`${edge.from}\u0000${edge.to}`, edge.cost])); const elements = [ // Pannable for the same reason as the union lane above. { data: { id: laneId, label: laneCostLabel(laneLabel, laneCost, stage), isLane: true }, classes: 'laneParent', selectable: false, grabbable: false, pannable: true }, @@ -657,7 +657,7 @@ function laneElements(laneId, laneLabel, graph, query, stage, laneCost) { data: { id: `${laneId}-${node.id}`, parent: laneId, - // On-graph label carries a concise cost/benefit badge (issue #286) + // On-DAG label carries a concise cost/benefit badge (issue #286) // when this node's decision has one; `node.label` itself (nested, // used everywhere else — the sidebar, schema derivation, …) stays // exactly the plain IR label. @@ -669,7 +669,7 @@ function laneElements(laneId, laneLabel, graph, query, stage, laneCost) { // object. kind: node.kind, category: categoryOf(node.kind), - root: node.id === graph.root, + root: node.id === dag.root, isPrePost: true, laneId, stage, @@ -687,7 +687,7 @@ function laneElements(laneId, laneLabel, graph, query, stage, laneCost) { elements.push({ data: { // Arrow points from input to consumer (data-flow direction), the - // reverse of the tree's parent->child structure. + // reverse of the DAG's parent->child structure. id: `e-${laneId}-${node.id}-${childId}`, source: `${laneId}-${childId}`, target: `${laneId}-${node.id}`, @@ -789,7 +789,7 @@ function renderCostAnnotation(title, annotation) { } // Baseline/selected/benefit trio for one replacement decision, matching -// `DagDecision.baseline_cost/selected_cost/benefit` (crates/types/src/dag_export.rs). +// `DAGDecision.baseline_cost/selected_cost/benefit` (crates/types/src/dag_export.rs). function renderDecisionCostBlock(entry) { if (!entry.baseline_cost && !entry.selected_cost && !entry.benefit) return ''; return `
@@ -800,8 +800,8 @@ function renderDecisionCostBlock(entry) {
`; } -// Short, on-graph badge text for a post-ASAP node's own winning decision — -// "concise on-graph benefit/cost badges" per issue #286; the full +// Short, on-DAG badge text for a post-ASAP node's own winning decision — +// "concise on-DAG benefit/cost badges" per issue #286; the full // breakdown only ever appears in the sidebar (`renderDecisionCostBlock`). // Empty string whenever there's nothing to show (no decision, or its // benefit is `Unavailable`) so an un-costed node's label is untouched. @@ -842,7 +842,7 @@ function computeSelectionWorkloadCost(selected) { let cacheProfile = null; let hasCacheProfile = false; for (const query of selected) { - const nodes = (query.post_graph && query.post_graph.nodes) || []; + const nodes = (query.post_dag && query.post_dag.nodes) || []; for (const node of nodes) { const decision = node.decision; if (!decision) continue; @@ -967,12 +967,12 @@ function showEdgeDetail(edge) { function renderScopeSummary(selected) { scopePickerEl.classList.add('visible'); const strategies = new Set(); - selected.forEach((query) => (query.post_graph?.nodes || []).forEach((node) => { + selected.forEach((query) => (query.post_dag?.nodes || []).forEach((node) => { if (node.decision && node.decision.rank === 0) strategies.add(node.decision.strategy); })); const scope = selected.length === 1 ? 'Single query' : `Batch workload · ${selected.length} queries`; const strategyText = strategies.size ? `Winning strategies: ${Array.from(strategies).join(', ')}` : 'No selected replacements'; - // Single query: use the exporter's own precomputed `NamedGraph.workload_cost` + // Single query: use the exporter's own precomputed `NamedDAG.workload_cost` // directly. Multiple queries: no single precomputed field covers exactly // this subset, so aggregate the explicit per-decision annotations already // in the export (dedup by `decision.id`) — see @@ -980,7 +980,7 @@ function renderScopeSummary(selected) { // client-side cost estimation. const costSummary = selected.length === 1 ? selected[0].workload_cost : computeSelectionWorkloadCost(selected); const benefitValue = costSummary && costSummary.benefit && costSummary.benefit.value; - const hasEdgeCosts = selected.some((query) => (query.post_graph?.edge_annotations || []).length > 0); + const hasEdgeCosts = selected.some((query) => (query.post_dag?.edge_annotations || []).length > 0); const outcomeLabel = typeof benefitValue !== 'number' ? 'difference' : benefitValue > 0 ? 'saved' : benefitValue < 0 ? 'additional cost' : 'no change'; @@ -1113,7 +1113,7 @@ function renderTableSchemas(qs) { const box = document.getElementById('tableSchemaBox'); const schemas = new Map(); qs.forEach((query) => { - (query.graph?.nodes || []).filter((node) => node.kind === 'Scan').forEach((node) => { + (query.dag?.nodes || []).filter((node) => node.kind === 'Scan').forEach((node) => { const key = JSON.stringify([node.detail, node.schema]); if (!schemas.has(key)) schemas.set(key, { node, owners: [] }); schemas.get(key).owners.push(query.name); @@ -1128,12 +1128,12 @@ function renderTableSchemas(qs) { function applyHighlighting() { if (!cy) return; const selected = getParticipants().map((i) => queries[i]); - const ownersFor = (graphOf) => { + const ownersFor = (dagOf) => { const owners = new Map(); selected.forEach((query) => { const seen = new Set(); - const graph = graphOf(query); - (graph ? graph.nodes : []).forEach((node) => { + const dag = dagOf(query); + (dag ? dag.nodes : []).forEach((node) => { const id = node.workload_node_id; if (id === undefined || seen.has(id)) return; seen.add(id); @@ -1143,8 +1143,8 @@ function applyHighlighting() { }); return owners; }; - const preOwners = ownersFor((query) => query.graph); - const postOwners = ownersFor((query) => query.post_graph); + const preOwners = ownersFor((query) => query.dag); + const postOwners = ownersFor((query) => query.post_dag); cy.nodes().forEach((element) => { if (element.data('isLane')) return; const node = element.data('node'); @@ -1174,7 +1174,7 @@ function renderLegend() { const panelBg = getComputedStyle(document.documentElement).getPropertyValue('--panel2').trim() || '#f0f2f5'; const mutedColor = getComputedStyle(document.documentElement).getPropertyValue('--muted').trim() || '#6b7280'; rows.push(`
- Pass-through (KeepPreAsap)Unchanged pre-ASAP subtree carried into the Summary graph as-is
`); + Pass-through (KeepPreAsap)Unchanged pre-ASAP sub-DAG carried into the Summary DAG as-is`); legendList.innerHTML = rows.join(''); } @@ -1203,13 +1203,13 @@ function loadWorkload(parsed) { const incoming = (parsed && parsed.queries) || []; // One batch id for this whole document — see `sourceBatch`'s own doc above. const sourceBatch = nextSourceBatch++; - incoming.forEach((q) => queries.push({ name: q.name, graph: q.graph, source: q.source, replacements: q.replacements || [], post_graph: q.post_graph, workload_cost: q.workload_cost, lifecycle_plan: q.lifecycle_plan, lifecycle_summary: q.lifecycle_summary, sourceBatch })); + incoming.forEach((q) => queries.push({ name: q.name, dag: q.dag, source: q.source, replacements: q.replacements || [], post_dag: q.post_dag, workload_cost: q.workload_cost, lifecycle_plan: q.lifecycle_plan, lifecycle_summary: q.lifecycle_summary, sourceBatch })); if (activeIndex === -1 && queries.length > 0) activeIndex = 0; if (participants.size === 0 && activeIndex >= 0) participants.add(activeIndex); } // Entry point used by planner-ui.js after the local HTTP backend returns a -// freshly generated WorkloadGraph. Replace (rather than append to) the +// freshly generated WorkloadDAG. Replace (rather than append to) the // current data and open the complete workload pre/post view. window.renderPlannerWorkload = function renderPlannerWorkload(parsed) { queries = []; @@ -1220,7 +1220,7 @@ window.renderPlannerWorkload = function renderPlannerWorkload(parsed) { render(); }; -// render.py's standalone output bakes the WorkloadGraph JSON directly into +// render.py's standalone output bakes the WorkloadDAG JSON directly into // the page as a